From 958441a9fc41832ad0a7ad1cc782befd1cc2d409 Mon Sep 17 00:00:00 2001 From: JUN Date: Thu, 24 Sep 2026 18:30:31 +0900 Subject: [PATCH 01/90] docs(plan): record 0.2.39 delivery (#249) --- .../260924_issue_sweep_0238/071_delivery.md | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) create mode 100644 devlog/_plan/260924_issue_sweep_0238/071_delivery.md diff --git a/devlog/_plan/260924_issue_sweep_0238/071_delivery.md b/devlog/_plan/260924_issue_sweep_0238/071_delivery.md new file mode 100644 index 00000000..196dd9c5 --- /dev/null +++ b/devlog/_plan/260924_issue_sweep_0238/071_delivery.md @@ -0,0 +1,16 @@ +# wp8 delivery: codexclaw 0.2.39 + +v0.2.39 is published and is the latest release: https://github.com/lidge-jun/codexclaw/releases/tag/v0.2.39 (stable, not a draft, published 2026-09-24T09:23:08Z). + +| Step | Evidence | +|---|---| +| PR to dev | #246, head b7846dcb, 14/14 checks success; squash-merged as 22ea08c5 | +| dev at 22ea08c5 | push CI 35976684535, Packed install 35976684536, WSL 35976684551 all success | +| Promotion | #248 (dev → main) green at 22ea08c5; merged with a merge commit as 8e6aa800586360f440b74baf18f91e8c8af5d659 | +| main | push CI 35979257804 and Packed install 35979257769 success | +| Release dry run | 35979974600: `release verify: READY — 0.2.39`, version kind stable, tests pass=3533 fail=0 total=3609 | +| Release publish | 35980577607 success; tag v0.2.39 → 8e6aa800 | +| Assets | `shasum -a 256 -c SHA256SUMS` OK; the payload equals `git archive 8e6aa800 plugins/codexclaw` (0 differing, 0 missing, 0 extra files); manifest `0.2.39+codex.20260924082502` | +| Issues | #243 closed as completed with an evidence comment; its unmet acceptance lines (a scan-free family signal and a live V2 run) are tracked in #247 | + +This run hit no flaky tests. From 693ead091f1857b166b6e1e17caf75e0f93a800d Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Sun, 27 Sep 2026 23:26:27 +0900 Subject: [PATCH 02/90] docs(plan): issue train 0927 roadmap (triage, wp2-wp4 diff-level docs) --- devlog/_plan/260927_issue_train/000_plan.md | 59 ++++ .../_plan/260927_issue_train/001_research.md | 35 ++ .../002_architect_consultation.md | 37 +++ .../010_wp2_runtime_overview.md | 51 +++ .../011_issue255_lazy_session_state.md | 106 +++++++ .../012_issue252_pabcd_switch.md | 83 +++++ .../013_issue253_idle_goal_release.md | 33 ++ .../014_issue254_turn_budget.md | 69 ++++ .../015_issue251_worker_gate.md | 37 +++ .../016_issue250_trigger_narrowing.md | 84 +++++ .../020_wp3_agent_thread_permissions.md | 285 +++++++++++++++++ .../021_wp3_dispatch_guidance.md | 245 ++++++++++++++ .../030_wp4_goalplan_decisions.md | 300 ++++++++++++++++++ .../260927_issue_train/040_wp5_delivery.md | 24 ++ 14 files changed, 1448 insertions(+) create mode 100644 devlog/_plan/260927_issue_train/000_plan.md create mode 100644 devlog/_plan/260927_issue_train/001_research.md create mode 100644 devlog/_plan/260927_issue_train/002_architect_consultation.md create mode 100644 devlog/_plan/260927_issue_train/010_wp2_runtime_overview.md create mode 100644 devlog/_plan/260927_issue_train/011_issue255_lazy_session_state.md create mode 100644 devlog/_plan/260927_issue_train/012_issue252_pabcd_switch.md create mode 100644 devlog/_plan/260927_issue_train/013_issue253_idle_goal_release.md create mode 100644 devlog/_plan/260927_issue_train/014_issue254_turn_budget.md create mode 100644 devlog/_plan/260927_issue_train/015_issue251_worker_gate.md create mode 100644 devlog/_plan/260927_issue_train/016_issue250_trigger_narrowing.md create mode 100644 devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md create mode 100644 devlog/_plan/260927_issue_train/021_wp3_dispatch_guidance.md create mode 100644 devlog/_plan/260927_issue_train/030_wp4_goalplan_decisions.md create mode 100644 devlog/_plan/260927_issue_train/040_wp5_delivery.md diff --git a/devlog/_plan/260927_issue_train/000_plan.md b/devlog/_plan/260927_issue_train/000_plan.md new file mode 100644 index 00000000..5e2c347e --- /dev/null +++ b/devlog/_plan/260927_issue_train/000_plan.md @@ -0,0 +1,59 @@ +# Issue train 2026-09-27: hook runtime fixes, agent-thread permissions, goalplan decisions + +Codexclaw has four confirmed defects in its hook runtime, one host workaround worth shipping, and three small opt-in improvements among the 22 open issues. The defects are: ordinary prompt words inject PABCD directives (#250), the SubagentStop evidence gate blocks Codex's built-in `worker` in sessions that never used PABCD (#251), the Stop hook blocks every session with an active native goal even when no goalplan is bound (#253), and SessionStart writes `.codexclaw/sessions/.json` into every working directory (#255). The host workaround covers threads that Codex Desktop creates through `create_thread` with reduced permission even when the user runs full access. This unit fixes the defects, adds the opt-in permission hook and advisory, adds a PABCD off switch (#252), a per-turn Stop budget (#254) and plan-local pending decisions (#262), and records a triage decision for every open issue (001). + +Reader: a maintainer deciding whether to merge these changes into dev; familiarity with the pabcd-state hook component and the goalplan CLI is assumed. + +## Loop contract + +- Loop archetype: satisfy-spec HOTL, docs-first (LOOP-DOCS-FIRST-01). +- Trigger: user request on 2026-09-27 to fix the real issues and worthwhile improvements among the open issues, plus the agent-thread permission problem found in the same chat, and put them into dev, through cxc-loop with unlimited gpt-6-sol dispatch. +- Goal: wp2-wp4 merged into `dev` through ordinary PRs after hosted CI; every open issue has a recorded decision; fixed issues are closed with PR links. +- Non-goals: dev to main promotion, release, version bump, tags, npm publish; Codex core or Desktop changes; other repositories; the deferred and declined proposals in 001. +- Verifier: per-phase focused `node --test` files named in each decade doc; at every C, `npm run build`, the focused tests through `cxc receipt test`, then full `npm test`, `node plugins/codexclaw/scripts/gate.mjs`, `node plugins/codexclaw/scripts/inventory.mjs --check --tests ` and `node plugins/codexclaw/scripts/platform-smoke.mjs`; hosted CI on each PR head (jobs actually ran, head SHA, event, run id). Skill prose changes are read by no test; their review is human (PLAN-VERIFIER-REAL-01). +- Stop condition: all seven goalplan criteria met with fresh evidence, or a real blocker after root-cause work. +- Memory artifact: this unit and `.codexclaw/evidence/01a0e313-aa5c-72a0-aecc-59a966bfca9c/`. +- Expected terminal outcomes: DONE (PRs merged, issues dispositioned); BLOCKED (CI infrastructure or branch protection outside scope); NEEDS_HUMAN (a default-on behavior change the user must choose); UNSAFE (a change would weaken a safety gate). +- Resource bounds: this checkout (`/Users/jun/.codex/worktrees/902d/codexclaw`), task-owned worktrees created with `create_worktree` for parallel builders, `gh` with the user's credentials. Writes limited to the IN scope below. No token or wall-clock bound was stated; host limits apply. + +## Scope and file map + +IN (details in each decade doc): + +``` +plugins/codexclaw/components/pabcd-state/{src,dist,test} wp2 (010-016), wp3 (020), wp4 (030) +plugins/codexclaw/components/cxc-ops/{src,dist,test} wp2 (011 SessionStart writes), wp3 (hook-trust) +plugins/codexclaw/components/recall, bg-wake (SessionStart only) wp2 (011) where a SessionStart write is found +plugins/codexclaw/hooks/*.json, .codex-plugin/plugin.json wp3 (two new hooks) +plugins/codexclaw/skills/{pabcd,loop,dev}/references/*.md wp2 (015, 012), wp3 (021), wp4 (030) +plugins/codexclaw/inventory.json, README family counts every phase that changes tests or hooks +``` + +OUT: Codex core/Desktop, host automation mutation handler (#213), SessionStart family detection from the host (#247), anything listed as defer or decline in 001. + +## Ordered work phases + +- wp1 (this cycle): docs-only roadmap: 001 triage, 002 architect consultation, 010-016, 020-021, 030, 040. No production edits. +- wp2: hook runtime fixes (010 overview; 011 #255 lazy session state, 012 #252 PABCD switch, 013 #253 IDLE goal release, 014 #254 per-turn Stop budget, 015 #251 worker gate, 016 #250 trigger narrowing). Foundation first: 011 changes how state comes to exist and 012 adds the switch every later PABCD handler consults; 013 and 014 then change Stop continuation; 015 and 016 are leaf policy changes. Built in parallel by gpt-6-sol builders on separate branches in task-owned worktrees, merged in that order. +- wp3: agent-created thread permissions (020 hook and advisory, 021 dispatch guidance plus #265 checkpoint guidance). After wp2 because the new SessionStart advisory must follow 011's no-write SessionStart rule and 012's switch semantics. +- wp4: goalplan pending decisions (030, #262). Independent of wp3; after wp2 so Stop/IDLE logic changes are settled before readiness semantics change. +- wp5: delivery and issue disposition (040). + +Delivery: one ordinary PR per implementation work phase from a `codex/issue-train-wpN` branch into `dev`, each merged after its hosted CI passes and the next phase rebased onto the new `dev`. No native stacks. Main owns git, the FSM, integration and delivery; gpt-6-sol subagents draft docs, build within disjoint scopes or task-owned worktrees, and review. + +## Issue acceptance mapping + +| Issue | Decision | Where it lands | +|---|---|---| +| #250 | fix | 016 | +| #251 | fix | 015 | +| #252 | implement (narrowed) | 012 | +| #253 | fix | 013 | +| #254 | implement | 014 | +| #255 | fix | 011 | +| #262 | implement | 030 | +| #265 | guidance only | 021 | +| agent-thread permissions (no issue) | implement | 020, 021 | +| #209 #213 #247 #256 #257 #258 #259 #260 #263 #264 #266 #267 #268 | defer | 001 | +| #261 | decline | 001 | + diff --git a/devlog/_plan/260927_issue_train/001_research.md b/devlog/_plan/260927_issue_train/001_research.md new file mode 100644 index 00000000..13ec79e0 --- /dev/null +++ b/devlog/_plan/260927_issue_train/001_research.md @@ -0,0 +1,35 @@ +# Triage of the 22 open issues (2026-09-27) + +Six gpt-6-sol explorers verified each issue against `dev` at `958441a9` (read-only, source anchors in their returns); main accepted their verdicts with the adjustments noted. REAL means the shipped behavior is wrong; PROPOSAL means the report asks for new behavior. + +| Issue | Verdict | Decision | Reason | +|---|---|---|---| +| #209 pending worktree thread has no clientThreadId to threadId path | NOT-REPRODUCED in plugin | defer | The gap is in the Desktop creation wrapper; codexclaw already keeps provisional and canonical ids apart (dispatch-surfaces.md:114-129, check-lane-packet.mjs:96-119). Needs a host completion event. | +| #213 automation ids are host-global | PARTIAL | defer | The ownership hook already denies foreign update/delete on hooked calls (automation-ownership-gate.ts:37-62). The remaining hole is atomic authorization inside the host mutation handler. | +| #247 identify collab family at SessionStart | PROPOSAL | defer | SessionStart has no tool catalog or family field (fallback-dispatch-cli.ts:33-39); a plugin-only fix would guess. Needs a host signal. | +| #250 trigger and loop-arm heuristics match ordinary words | REAL | fix (016) | detectTrigger and detectLoopArmRequest match bare mentions and generic persistence phrases (hook.ts:238-285) and inject directives into headless runs. | +| #251 SubagentStop gate blocks built-in worker | REAL | fix (015) | The gate checks agent type without checking whether the parent armed a PABCD cycle (subagent-evidence.ts:471). | +| #252 no supported PABCD off switch | PROPOSAL | implement narrowed (012) | A blanket pabcd-state no-op would also remove worktree, memory-write and automation guards (inventory.json:175-205); a PABCD-policy switch keeps them. | +| #253 GOAL-IDLE-CONTINUE-01 blocks every active-goal session | REAL | fix (013) | handleStop blocks at IDLE whenever a native goal is active, even with no state or goalplan (hook.ts:1781-1787; hook-continuation.test.ts:506 pins it). | +| #254 MAX_STOP_BLOCKS_TOTAL never resets | PROPOSAL (behavior intended) | implement (014) | The cap is documented as per-session (hook.ts:1365); long goal sessions still lose continuation silently. A per-real-user-turn budget keeps the unattended bound. | +| #255 SessionStart writes session state into every cwd | REAL | fix (011) | handleSessionStart calls ensureState unconditionally (hook.ts:572, state.ts:316-368). | +| #256 independent verification receipt | PROPOSAL | defer | Receipts store a joined command string with no argv, cwd or output digests (receipt-cli.ts:170-172); a rerun cannot be reconstructed reliably yet. | +| #257 strict accepted-progress report | PROPOSAL | defer | Review rounds attach to plan audit, not completed work (review-round-cli.ts:230-235); needs a post-implementation acceptance record first (#256). | +| #258 verbatim archive of user-typed prompts | PROPOSAL | defer | UserPromptSubmit carries no typed-versus-injected provenance (hook.ts:134, parse.ts:65); the archive cannot promise what it claims. | +| #259 provenance on peer prompts | PROPOSAL | defer | An unauthenticated text envelope that suppresses PABCD parsing would let anyone type it; needs authenticated sender metadata from the host. | +| #260 unit fields and cxc loop check | PROPOSAL | defer | Fields would be dropped by the reviver today (goalplan.ts:548); the external-wait part is covered by #262's decision links. | +| #261 reviewer panel per round | PROPOSAL | decline | Turning lanes and synthesis into gates reverses the deliberate non-blocking review round (orchestrate-cli.ts:65); parallel reviewers already work under guidance. | +| #262 pending decisions in goalplans | PROPOSAL | implement (030) | Opt-in plan-local record; readiness derives waits from open decisions. | +| #263 PreCompact checkpoint hook | PROPOSAL | defer | Codex supports PreCompact (hooks/src/schema.rs:347), but the proposal adds several stores; revisit after #255 settles cwd writes. | +| #264 measured-state snapshot after compaction | PROPOSAL | defer | Readers map absence to defaults (state.ts:486) and bg listing writes corrections (registry.ts:110-154); needs a strict read contract first. | +| #265 on-disk progress checkpoint for workers | PROPOSAL | guidance only (021) | Useful as a packet convention; a new CLI verb is not needed yet. | +| #266 host hygiene doctor | PROPOSAL | defer | Process ownership across platforms is not establishable; bg reconcile writes (registry.ts:106-146). | +| #267 suggested wave width | PROPOSAL | defer | Family, native limit and open-child count are not observable at SessionStart (dispatch-card.ts:24-36). | +| #268 compact-at-clean-boundary advisory | PROPOSAL | defer | No token usage parsing exists and Stop systemMessage display is unverified (stop.rs:288). | + +Declined and deferred issues stay open with a comment linking this record, except #261, which is closed as declined. + +## Agent-created thread permissions (no issue) + +Measured on this Mac on 2026-09-27. Rollouts with `thread_source=agent_created_thread` started with `approval_policy=on-request` and `workspace-write` in 7 of about 356 September cases (plus 8 archived on 09-23), including a projectless thread on 09-27, while their parents ran `never` with `danger-full-access`. The Desktop per-thread store held 5 `:workspace` entries out of 1,159. The only command-approval responses in the retained Desktop logs (05:06 and 08:53 UTC on 09-27) came from such children; thread `01a0e145` contains the exact `git fetch origin dev --quiet` command approved at 08:53. Subagents followed their parent in every case (5,336 never to never, 40 on-request to on-request). Codex runs PermissionRequest hooks before the user approval UI (codex-rs/core/src/tools/approvals.rs:505-525), and the bundled runtime 0.158.0-alpha.2.1 contains that path. Upstream: openai/codex #33282, #40793, #41167. + diff --git a/devlog/_plan/260927_issue_train/002_architect_consultation.md b/devlog/_plan/260927_issue_train/002_architect_consultation.md new file mode 100644 index 00000000..df25465f --- /dev/null +++ b/devlog/_plan/260927_issue_train/002_architect_consultation.md @@ -0,0 +1,37 @@ +# Architect consultation record + +Handle: `01a0e32b-c89d-7901-8a08-4af390a8081d` (gpt-6-sol, CXC-ROLE: architect, read-only), dispatched 2026-09-27 for the agent-thread permission design. The same handle performs the reflection check on the submitted roadmap. + +## Proposal and main dispositions + +| Decision | Summary | Main disposition | +|---|---|---| +| AD-1 | New module `pabcd-state/src/agent-thread-permissions.ts`, two hook verbs dispatched before the subagent early exit; allow or no decision, never deny | Accepted | +| AD-2 | Opt-in `permissions.agentCreatedThreadAutoAllow` in user-global `$CODEXCLAW_HOME/config.json`; project files cannot enable it | Accepted | +| AD-3 | Eligibility from the first bounded line of `transcript_path` (session_meta id match, thread_source agent_created_thread), `permission_mode=default`, no agent fields, top-level never plus danger-full-access in CODEX_HOME config, any profile means unknown | Accepted; forked threads stay out until measured | +| AD-4 | Matcher `*`, self-filter Bash, write_stdin, apply_patch, request_permissions and `mcp__` names; exact allow JSON | Accepted | +| AD-5 | SessionStart advisory regardless of opt-in, claims only degraded approval mode | Accepted; must obey 011's no-write SessionStart rule | +| AD-6 | Guidance: create_worktree plus a subagent that passes the worktree as shell workdir; threads stay the surface for lanes needing their own goal | Accepted | +| AD-7 | Bypass record: C4, PermissionRequest hook, opt-in allow suppresses user approval, residual risks listed | Accepted | + +Rejected alternatives recorded by the architect: rewriting the approval policy at startup, enabling from a repository file, treating unknown tool names as MCP, trusting config text as runtime proof, claiming a subagent's native cwd changes. + +Upstream gap: complete coverage needs Codex to expose the resolved sandbox state and an approval action kind in PermissionRequest input. The shipped hook is scoped to what the payload proves today. + +## Reflection + +## Main dispositions of builder-doc questions + +- 011: a bound `cxc loop init --session cli` keeps the existing standalone-terminal exception, matching the reserved `cli` key in orchestrate; every other bound id requires native verification before any write. +- 030: the decision record stays the smaller `open|decided` contract. Issue #262's `options[]` and `withdrawn` are not adopted; a withdrawn question is decided with an answer that says so. + + +## Reflection result (same handle, 2026-09-27) + +Verdict: MISALIGNED on AD-3 wording and AD-4 MCP scope; AD-1, AD-2, AD-5, AD-6, AD-7 ALIGNED. Gaps and main dispositions: + +1. The success claim said the hook proves full access; it only sees top-level config evidence and a tool-name convention. Folded: 020's success line now says "explicit top-level config evidence" and names the `mcp____` convention, and states the exact guarantee needs upstream fields. +2. Cross-phase invariants were unpinned. Folded: both permission verbs stay active under `CODEXCLAW_PABCD=off` (020 dependency line, 012 note), and 020 requires a built-CLI integration test with the switch off and a fresh cwd that asserts output and no `/.codexclaw`. The CLI hook-observation write goes to `CODEX_HOME`, which 011's rule allows. +3. Build and inventory order. Folded: build first, then focused tests, then full `npm test`, then `inventory.mjs --write/--check --tests ` and gate. + +No further module-ownership conflicts were found across 010-016, 020 and 030. diff --git a/devlog/_plan/260927_issue_train/010_wp2_runtime_overview.md b/devlog/_plan/260927_issue_train/010_wp2_runtime_overview.md new file mode 100644 index 00000000..ee6e7390 --- /dev/null +++ b/devlog/_plan/260927_issue_train/010_wp2_runtime_overview.md @@ -0,0 +1,51 @@ +# wp2 — Runtime issue train (#255, #252, #253, #254, #251, #250) + +This work phase makes idle Codex sessions passive, provides a PABCD off switch, and narrows continuation and delegation gates to the sessions they govern. Six independent builder branches should land in this order: `011` lazy state, `012` switch, `013` idle release, `014` turn budget, `015` worker gate, `016` trigger narrowing. Rebase each later branch on the just merged predecessor, rerun its targeted tests, then run the complete phase gate. All anchors below refer to branch `codex/issue-train-0927` at `958441a9`; each builder must recheck them after predecessor merges. + +## Phase contract + +- `011` owns first-write identity and the SessionStart audit. It must land first because every later hook can encounter a missing state file. +- `012` owns the off switch in hook dispatch. The safety guards remain active; see its explicit allowlist. +- `013` makes active but unbound goals release at IDLE without writing counters. +- `014` makes the absolute Stop cap apply to one real user turn. Native Stop continuations do not create new UserPromptSubmit inputs: `/tmp/cxc-perm/codex-src/codex-rs/core/src/session/turn.rs:666-683`, `/tmp/cxc-perm/codex-src/codex-rs/core/src/hook_runtime.rs:682-707`. +- `015` gates registered executor unconditionally and legacy worker only under an armed PABCD cycle. +- `016` leaves explicit command parsing unchanged and eliminates incidental natural-language arming. + +## Shared-file merge map + +| Path | Planned owners | Conflict resolution | +| --- | --- | --- | +| `plugins/codexclaw/components/pabcd-state/src/hook.ts` | 011, 013, 014, 016 | Preserve 011's missing-state behavior; apply 013 around `handleStop` guard 2a, 014 around `handleUserPromptSubmit` and `bumpStopCounter`, then 016 detector replacements. | +| `plugins/codexclaw/components/pabcd-state/src/cli.ts` | 012, possibly 011 | Keep 012 dispatch guard above the PABCD-only branches while preserving 011's mutating CLI identity checks. | +| `plugins/codexclaw/components/pabcd-state/src/state.ts` | 011, 014 | Keep `ensureState` exclusive-create contract and add 014's `stopBlockTurnId` to `State`, default, and strict reconstruction. | +| `plugins/codexclaw/components/pabcd-state/test/hook*.test.ts` | 011–016 | Merge by named tests; update existing expectations instead of retaining contradictory tests. | + +No builder may overwrite another branch's full file. Resolve conflicts in the listed order; run the named tests after each merge. The PABCD hook dispatch is `cli.ts:329-488`; state serialization is `state.ts:486-615`; test globs are in root `package.json:24`. + +## Verifier reality check (PLAN-VERIFIER-REAL-01) + +These commands were run before the plan was written on this HEAD: + +| Command | Exit | Observes | +| --- | ---: | --- | +| `node --test plugins/codexclaw/components/pabcd-state/test/state.test.ts plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts plugins/codexclaw/components/pabcd-state/test/subagent-evidence.test.ts` | 0; 153 pass, 0 fail | Existing state, Stop, and evidence test files directly. It cannot prove any new tests or production behavior. | +| `node plugins/codexclaw/scripts/inventory.mjs --check` | 0; 29 skills, 29 hooks, 9 components | Package inventory, not the behavior or prose in these seven docs (`inventory.mjs:356-365`). | + +The builders run `node --test ` after each fix, then `npm run build` to regenerate committed `dist` from changed sources (`package.json:22`), `npm test` for the root suite (`package.json:24`), `npm run gate` for repository gate checks (`package.json:23`), and `node plugins/codexclaw/scripts/inventory.mjs --check --tests ` using the actual full-suite count (`inventory.mjs:416-421`, `.github/workflows/ci.yml:62`). `npm run build` was deliberately not run during this docs-only delegation because it writes outside the seven authorized files. The builder must inspect the resulting dist diff and update any published test count only from the measured suite, never a guessed increment. Human review must check the policy prose; inventory and the gate do not validate this plan's reasoning. + +## Activation and boundary matrix + +| Scenario | Expected | +| --- | --- | +| Fresh `SessionStart`, no `.codexclaw` | No directory or state file created (`011`). | +| First verified mutating command | Exclusive state creation; failed native verification leaves no state (`011`). | +| `CODEXCLAW_PABCD=off` or project `pabcd.enabled=false` | PABCD hooks silent; independent safety guards still run (`012`). | +| Active host goal, no bound plan, IDLE Stop | Release without counter write (`013`). | +| Bound goal, IDLE Stop | Existing bounded arming behavior (`013`). | +| 25 Stop continuations in one user turn | Release on the 25th, with one nonblocking message; a later real user turn gets a fresh total (`014`). | +| Plain worker without armed cycle | Release without attempt/tombstone; executor or armed worker uses receipt gate (`015`). | +| Incidental interview/build/verify/persistence wording | No trigger, context, or `loopArmSeen` write; explicit CodexClaw request still works (`016`). | + +## Out of scope + +No relocation to `CODEXCLAW_HOME`, blanket `.gitignore`, change to explicit `orchestrate` parser grammar, new hook registration, or relaxation of worktree/memory/automation safety gates. No source or test change is made in this planning pass. diff --git a/devlog/_plan/260927_issue_train/011_issue255_lazy_session_state.md b/devlog/_plan/260927_issue_train/011_issue255_lazy_session_state.md new file mode 100644 index 00000000..17b51110 --- /dev/null +++ b/devlog/_plan/260927_issue_train/011_issue255_lazy_session_state.md @@ -0,0 +1,106 @@ +# #255 — Create session state on the first verified mutation + +`SessionStart` must leave a new repository without `.codexclaw`. The first `cxc session bind`, `cxc orchestrate --session `, or `cxc loop init --session ` creates the state after the same native identity check. Reads and failed verification remain read-only. + +Current anchors: `plugins/codexclaw/components/pabcd-state/src/hook.ts:572`, `plugins/codexclaw/components/pabcd-state/src/state.ts:368`, `plugins/codexclaw/components/pabcd-state/src/session-cli.ts:76`, `plugins/codexclaw/components/pabcd-state/src/orchestrate-cli.ts:538`, `plugins/codexclaw/components/pabcd-state/src/goalplan-cli.ts:675`. + +## File change map + +### MODIFY `plugins/codexclaw/components/pabcd-state/src/hook.ts` + +At `handleSessionStart` (`:567-576`), replace `ensureState(payload.cwd, payload.session_id)` with a read-only existing-file check, or simply remove it: the handler has no refresh work today. Keep its empty output. `ensureState` at `state.ts:368-410` creates `.codexclaw/sessions`; `readState` supplies a virtual default without writing (`state.ts:486-603`). Do not call `writeState` from SessionStart. If the hook needs to refresh an existing file in a later change, check existence and preserve its contents. + +```diff + export function handleSessionStart(payload: SessionStartPayload): string { + if (payload.hook_event_name !== "SessionStart") return ""; +- ensureState(payload.cwd, payload.session_id); + return ""; + } +``` + +Also protect ordinary UserPromptSubmit from minting state: `hook.ts:643-653` currently writes a memory-request marker before it knows whether a state file exists, and `:698-705` writes `loopArmSeen`. At `:630`, compute `const stateExists = sessionStateFileExists(payload.cwd, payload.session_id)` using the new read-only helper below. Guard each `writeState` in this handler with `stateExists`; on a missing file, allow context-only guidance but never persist `injectedTurns`, `loopArmSeen`, or the memory marker. A memory request still has to be enforced separately by the independent PreToolUse memory gate; the no-state case must deny rather than silently authorize. Add a test that a fresh normal prompt and a fresh loop-arm prompt leave `.codexclaw` absent. An explicit chat `orchestrate` command is a mutating hook path at `hook.ts:659-662`; on missing state it must emit a “run cxc session bind or verified cxc orchestrate” instruction and avoid calling `handleOrchestrateCommand`, which otherwise writes an unverified state. + +### MODIFY `plugins/codexclaw/components/pabcd-state/src/state.ts` + +Expose a read-only existence helper beside private `statePath` (`:320-322`); `existsSync` is already imported. This prevents callers from accidentally defaulting a missing file into a write. + +```ts +export function sessionStateFileExists(cwd: string, sessionId: string): boolean { + return isCanonicalSessionId(sessionId) && existsSync(statePath(cwd, sessionId)); +} +``` + +Do not change the exclusive-create `ensureState` algorithm (`:368-410`). + +### MODIFY `plugins/codexclaw/components/pabcd-state/src/orchestrate-cli.ts` + +`runOrchestrateCli` requires explicit `--session` (`:519-535`) and currently rejects a missing file (`:538-548`). Replace only that missing-file branch. Reuse `resolveNativeSession(args.cwd, nativeEnv)` from `session-binding.ts:41-109`; require `native.ok`, exact `native.sessionId === sessionId`, and exact native cwd. Then call `ensureState(native.cwd, native.sessionId)` and inspect/read it. A reserved standalone `cli` key keeps its existing behavior. A mismatch, absent `CODEX_THREAD_ID`, missing/archived/subagent native row, wrong cwd, or malformed existing state fails before any mutation. `session-cli.ts:76-91` is the reference for verify → exclusive create → inspect. Status at `orchestrate-cli.ts:499-517` remains read-only. + +Replace `orchestrate-cli.ts:543-548` with this complete block: + +```ts +if (args.session && !sessionFileExists(args.cwd, sessionId) && !RESERVED_SESSION_KEYS.has(sessionId)) { + const native = resolveNativeSession(args.cwd, nativeEnv); + if (!native.ok || native.sessionId !== sessionId) { + return { code: 1, output: `orchestrate ${args.verb}: native session verification failed; nothing was written` }; + } + try { ensureState(native.cwd, native.sessionId); } + catch { return { code: 1, output: `orchestrate ${args.verb}: could not create session state; nothing was written` }; } + const checked = inspectState(native.cwd, native.sessionId); + if (!checked.ok || !checked.stateExists) { + return { code: 1, output: `orchestrate ${args.verb}: session state invalid after creation; nothing was written` }; + } +} +``` + +`inspectState` is currently private (`session-cli.ts:13`); export it and import it into `orchestrate-cli.ts`, along with `ensureState`. This reuses the raw regular-file/symlink and session-id checks that `session bind` performs. Do not invent a second native verifier. Preserve `resolveSessionSource` before phase writes (`orchestrate-cli.ts:568-572`). + +### MODIFY `plugins/codexclaw/components/pabcd-state/src/session-cli.ts` + +Change `function inspectState` at `:13` to `export function inspectState` without changing its body. `session bind` already calls it before and after exclusive creation (`:76-91`). + +### MODIFY `plugins/codexclaw/components/pabcd-state/src/goalplan-cli.ts` + +The bound `init` path checks source identity at `:669-679`, then writes the plan and finally writes default session state at `:681-699`. Move native verification plus exclusive state creation before `writeGoalplan` so rejected identities leave neither plan nor state. Use the same `resolveNativeSession` contract and exact session/cwd match; skip this step for unbound `init` to preserve its local-artifact behavior (`:672-674`). After creation, preserve source identity checking and the existing slug binding. The `runGoalplanCli` signature needs injectable `nativeEnv: NodeJS.ProcessEnv = process.env` for focused tests, or a shared verified-create helper that both CLIs call. A helper belongs in `session-binding.ts`, which already owns native verification; no copy of SQLite logic. + +Change the `runGoalplanCli` signature at `:651` to `runGoalplanCli(args: GoalplanCliArgs, nativeEnv: NodeJS.ProcessEnv = process.env)`. Replace the bound-session check block at `:675-680` with this complete block, leaving `writeGoalplan` at `:686` and slug binding at `:695-698` in place: + +```ts +if (typeof args.session === "string" && args.session.length > 0) { + const native = resolveNativeSession(args.cwd, nativeEnv); + if (!native.ok || native.sessionId !== args.session) { + return { output: "loop init: native session verification failed; nothing was written", code: 1 }; + } + const before = inspectState(native.cwd, native.sessionId); + if (!before.ok) return { output: `loop init: ${before.error}; nothing was written`, code: 1 }; + const gate = checkBoundSourceIdentity(args.cwd, args.session); + if (!gate.ok) return { output: `loop init: ${gate.reason}\nNothing was written.`, code: 1 }; + try { ensureState(native.cwd, native.sessionId); } + catch { return { output: "loop init: could not create session state; nothing was written", code: 1 }; } + const after = inspectState(native.cwd, native.sessionId); + if (!after.ok || !after.stateExists) { + return { output: "loop init: state invalid after creation; nothing was written", code: 1 }; + } +} +``` + +Import `resolveNativeSession` from `session-binding.ts`, `inspectState` from `session-cli.ts`, and `ensureState` from `state.ts` at `goalplan-cli.ts:40-48`. Existing `cli.ts:171` passes no second argument and therefore uses the live environment. Any synthetic-ID `loop init --session` tests must create a native fixture or expect rejection; only unbound init retains its old no-native path. + +### READ ONLY `plugins/codexclaw/components/cxc-ops/src/map-affordance.ts` + +`runMapAffordanceSessionStart` is context-only (`:296-337`) and does not call `recoveryPath`. `recoveryPath` can create `.codexclaw/affordance-recovery` at `:241-249` only for `runPostCompactAffordance` (`:257-265`), not SessionStart. Keep this distinction and add a regression check; no production diff needed unless the branch discovers another SessionStart call to `recoveryPath(..., true)`. + +### No production change: recall and bg-wake + +Recall SessionStart builds context at `recall/src/hook.ts:699-737`; its index writes are under the user's home, not the cwd `.codexclaw` (`recall/src/index-db.ts:5-22,93`). `bg-wake/src/hook.ts:114-127` calls orphan adoption. `bg-wake/src/registry.ts:236-249` only writes when `listRecords` finds an existing terminal undelivered record, so a cwd with no `.codexclaw` creates nothing; retain adoption because it preserves completed background work across restart. Assert empty-directory behavior in tests. `cxc-ops` PostCompact marker is a separate event and may still create a recovery directory (`map-affordance.ts:257-265`). + +### MODIFY tests + +- `plugins/codexclaw/components/pabcd-state/test/state.test.ts:27-68`: rename the test to `ensureState: first verified mutation creates exact default state`; keep its exact default/temporary-file assertions. Add `SessionStart: fresh cwd remains without .codexclaw`, call `handleSessionStart`, assert empty output and `existsSync(join(cwd, STATE_DIR)) === false`; this fails before the fix. +- `plugins/codexclaw/components/pabcd-state/test/orchestrate-cli.test.ts`: `missing native state is created by matching verified orchestrate mutation`; `unknown explicit id or wrong cwd leaves .codexclaw absent`; `status missing state stays read-only`. Each must inspect exact state file and exit code. Existing native fixture helpers should be reused. +- `plugins/codexclaw/components/pabcd-state/test/goalplan.test.ts`: `bound loop init verifies native identity before plan/state writes`; assert wrong ID leaves both paths absent and matching ID creates both. This fails today because the bound path writes a default state without native verification. +- `plugins/codexclaw/components/cxc-ops/test/map-affordance.test.ts`, `plugins/codexclaw/components/bg-wake/test/hook.test.ts`: call SessionStart in an empty cwd and assert no `.codexclaw` exists. + +## Activation and bypass record + +Fresh SessionStart, resumed SessionStart with an existing state, fresh prompt, first valid mutation, invalid native ID/cwd/source, two concurrent creators, and empty bg/recall/map hooks must each be exercised. Native verification tier: CLI boundary; executing surface: `session bind`, orchestrate, bound loop init; known bypass: a direct library caller can still call `writeState`, and hook payloads are not independently authenticated; residual risk: same-user tampering with the native DB/env; wording: “verified CLI first mutation,” not a universal security boundary. Final enforcement layer is each mutating CLI entry. No relocation of `.codexclaw`, blanket ignore rule, or new SessionStart write. diff --git a/devlog/_plan/260927_issue_train/012_issue252_pabcd_switch.md b/devlog/_plan/260927_issue_train/012_issue252_pabcd_switch.md new file mode 100644 index 00000000..49177917 --- /dev/null +++ b/devlog/_plan/260927_issue_train/012_issue252_pabcd_switch.md @@ -0,0 +1,83 @@ +# #252 — PABCD hook policy switch + +`CODEXCLAW_PABCD=off` disables PABCD hook behavior for the process. Otherwise a project root `codexclaw.json` with `{ "pabcd": { "enabled": false } }` disables it. Missing, malformed, or other values default to enabled. The environment setting wins over the project setting. This changes hook dispatch only; it does not erase state or disable CLI commands. + +Current anchors: `plugins/codexclaw/components/pabcd-state/src/cli.ts:329`, `plugins/codexclaw/components/pabcd-state/src/interview-policy.ts:26`, `plugins/codexclaw/components/pabcd-state/src/goal-gate.ts:313`, `docs-site/src/content/docs/guides/pabcd.md:67`. + +## File change map + +### MODIFY `plugins/codexclaw/components/pabcd-state/src/interview-policy.ts` + +This module already owns the project config filename and fail-safe JSON read (`:26-61`). Add `readPabcdEnabled(cwd: string, env: NodeJS.ProcessEnv = process.env): boolean` beside `readInterviewPolicy`. Use `configPath(cwd)`; parse only a plain object with a plain-object `pabcd` member and boolean `enabled`. The only disabling values are exact env `off` (after `trim().toLowerCase()`) and exact JSON boolean `false`. No write is needed: `writeInterviewPolicy` preserves unrelated keys (`:63-95`). + +```ts +export function readPabcdEnabled(cwd: string, env: NodeJS.ProcessEnv = process.env): boolean { + if (env.CODEXCLAW_PABCD?.trim().toLowerCase() === "off") return false; + try { + const raw: unknown = JSON.parse(readFileSync(configPath(cwd), "utf8")); + if (!raw || typeof raw !== "object" || Array.isArray(raw)) return true; + const pabcd = (raw as Record).pabcd; + if (!pabcd || typeof pabcd !== "object" || Array.isArray(pabcd)) return true; + return (pabcd as Record).enabled !== false; + } catch { return true; } +} +``` + +### MODIFY `plugins/codexclaw/components/pabcd-state/src/cli.ts` + +After raw hook observation at `:333-341` and before PABCD handlers (`:391-480`), parse only `cwd` from the hook payload, fall back to `process.cwd()` for malformed input, and apply the explicit allowlist. Keep the three independent safety branches above it (`worktree-guard-pretool :357-364`, `pre-tool-use-memory-write :374-381`, `pre-tool-use-automation-ownership :383-387`). The subagent early return at `:391-393` can stay above the new switch; `subagent-stop` is deliberately exempt there. For a disabled event, exit 0 without output or state writes. No synthetic allow response is needed: an empty hook response means no intervention. + +```ts +const PABCD_DISABLED_EVENTS = new Set([ + "session-start", "user-prompt-submit", "stop", "post-compact", + "post-tool-use", "subagent-stop", "subagent-stop-review", + "pre-tool-use-idle-edit", "pre-tool-use-friction", "post-tool-use-friction", + "post-tool-use-edit-shape", "post-tool-use-render-observation", +]); +``` + +Insert this exact dispatch fragment after the subagent early exit at `cli.ts:391-393` and before the `pre-tool-use` branch at `:395-402`: + +```ts +let hookCwd = process.cwd(); +try { + const payload: unknown = JSON.parse(raw); + if (payload && typeof payload === "object" && !Array.isArray(payload)) { + const candidate = (payload as Record).cwd; + if (typeof candidate === "string" && candidate.length > 0) hookCwd = candidate; + } +} catch { /* malformed hook input keeps process cwd */ } +const pabcdEnabled = readPabcdEnabled(hookCwd); +if (!pabcdEnabled && PABCD_DISABLED_EVENTS.has(event)) process.exit(0); +``` + +Import `readPabcdEnabled` from `interview-policy.ts` at `cli.ts:23-53`. Use `pabcdEnabled` in the mixed branch: replace `if (output === "") output = handleIdleEditAdvisory(raw)` at `:442` with `if (pabcdEnabled && output === "") output = handleIdleEditAdvisory(raw)`. This leaves lint active. `pre-tool-use` stays live and is split in `goal-gate.ts` below. + +Split mixed branches, rather than skipping independent logic: `pre-tool-use-edit :438-442` must still execute `handleApplyPatchLint` but must skip `handleIdleEditAdvisory`; `post-tool-use-edit-shape :459-467` can no-op because both its shape hint and render capture serve PABCD; `pre-tool-use-lint :433-436` always stays. `session-start-rules :478-480` and `worktree-guard :451-454` stay, because project rules and worktree identity are not PABCD policy. `pre-tool-use :395-402` stays for goal-complete and goal-budget safety, but inspect `goal-gate.ts` branches: only `request_user_input` Interview/goal-mode prohibition is PABCD-specific and should return no intervention under the switch. Goal completion, evidence tombstones and budget protection are independent host-goal safety. Do not change recall's separate component hooks. + +The switch must cover `SubagentStop` evidence for both executor and worker when disabled; it does not delete old attempts or tombstones. A standalone `subagent-stop-review` observer is PABCD audit state and no-ops. Existing safety hooks registered outside this component remain untouched. + +### MODIFY `plugins/codexclaw/components/pabcd-state/src/goal-gate.ts` + +At `handlePreToolUseFailClosed` (`:313-321`), parse the payload, compute `const enabled = readPabcdEnabled(payload.cwd)`, then return `applyGoalBudgetGuard(payload) || (enabled ? applyGoalModeInterviewGuard(payload, deps) : "") || applyGoalCompleteGuard(payload, enabled)`. In the catch, env-off must not trigger the `request_user_input` fail-closed fallback; for a malformed payload use `readPabcdEnabled(process.cwd())` before that fallback. This is necessary because the `pre-tool-use` dispatcher remains live. Test enabled/disabled branches with goal active and inactive. + +`applyGoalBudgetGuard` checks the independent `create_goal` input shape (`goal-gate.ts:104-124`) and always stays. `applyGoalCompleteGuard` mixes boundaries (`:209-299`): add optional `pabcdEnabled = true`. When disabled, skip only the in-flight PABCD phase denial at `:226-229` and bound-goalplan quality denial at `:275-296`. Keep unreadable-state denial and unresolved subagent verdict/tombstone checks at `:215-225,231-274`, because existing evidence debt must not disappear when policy turns off. This is the explicit decision for goal-complete; it cannot remain byte-identical because two of its predicates depend on PABCD. Test that a previously in-flight but otherwise clean state can complete with PABCD off, while an unresolved tombstone still denies. Goal-budget has no PABCD dependency and stays intact. + +### MODIFY `docs-site/src/content/docs/guides/pabcd.md` + +Add a short “Disable PABCD hooks” subsection near the existing runtime lifecycle guidance (`:67`). Show both exact forms, env precedence, defaults, and the retained safety guards. The present project config reader is `interview-policy.ts:26-61`; the new `pabcd` key belongs in that same `codexclaw.json`. State that `cxc config interview off` only changes Interview promotion (`cli.ts:209-245`), whereas this switch disables PABCD hook dispatch. Do not claim the switch disables the CLI. + +### MODIFY tests + +- `plugins/codexclaw/components/pabcd-state/test/interview-policy.test.ts`: `pabcd switch: env off overrides project enabled`; `project false disables, malformed and missing enable`; `interview policy write preserves pabcd key`. Assert booleans and preserved JSON. +- `plugins/codexclaw/test/hook-e2e.test.mjs`: `pabcd off silences UserPromptSubmit, Stop, SessionStart, PostCompact, SubagentStop`; invoke the built hook entry with project config/env, assert stdout empty and no new session/attempt file. This fails before the switch because trigger and Stop hooks still emit/write. +- Same e2e file: `pabcd off retains worktree, memory, automation and apply_patch lint guards`; feed each registered event an existing deny fixture and assert its denial survives. `pre-tool-use-edit` must still deny a lint violation and emit no idle advisory. +- `plugins/codexclaw/components/pabcd-state/test/goal-gate.test.ts`: `pabcd off allows request_user_input while preserving goal completion and budget denials`. + +## Activation and bypass record + +Exercise env off with project true, env unset with project false, env `ON` with project false, invalid JSON, missing file, root/subagent payloads, and every mixed event. Tier: local hook dispatch policy, not a host-wide guarantee. Executing surface: `pabcd-state` hook CLI. Known bypass: direct library calls, terminal CLI commands, and an uninstalled/untrusted hook; residual risk: other components may issue independent context. Wording downgrade: “PABCD hooks in this component are silent,” not “CodexClaw is disabled.” Final enforcement layer: `cli.ts` hook dispatch plus `goal-gate.ts`'s Interview branch. Out of scope: recall hooks, state deletion, CLI write blocking, and safety-guard disablement. + +## Note from the roadmap reflection + +The wp3 hook verbs `permission-request` and `session-start-permission-advisory` (020) are not PABCD policy. The switch must not list them, and wp3's integration test asserts they still run with `CODEXCLAW_PABCD=off`. diff --git a/devlog/_plan/260927_issue_train/013_issue253_idle_goal_release.md b/devlog/_plan/260927_issue_train/013_issue253_idle_goal_release.md new file mode 100644 index 00000000..b50e309c --- /dev/null +++ b/devlog/_plan/260927_issue_train/013_issue253_idle_goal_release.md @@ -0,0 +1,33 @@ +# #253 — Release an unbound active goal at IDLE + +An active host goal alone must not make an IDLE PABCD session block Stop. The IDLE arming block applies only when `state.slug` resolves to a bound goalplan. A missing state or empty slug releases without creating state or advancing a counter. + +Current anchors: `plugins/codexclaw/components/pabcd-state/src/hook.ts:1392`, `plugins/codexclaw/components/pabcd-state/src/hook.ts:1787`, `plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts:506`, `plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts:567`. + +## File change map + +### MODIFY `plugins/codexclaw/components/pabcd-state/src/hook.ts` + +`handleStop` reads virtual state at `:1774`, determines `goalActive` and `inFlight` at `:1781-1782`, then calls `bumpStopCounter` for every active IDLE goal at `:1787-1793`. `safeReadBoundGoalplan` (`:1392-1396`) is the existing path-safe reader. Put its check before context-pressure inspection and counter write. This treats a stale/nonexistent slug as unbound rather than generating a false block. + +```diff + if (!inFlight) { + if (!goalActive) return ""; ++ if (!state.slug || !safeReadBoundGoalplan(payload.cwd, state.slug)) return ""; + if (isContextPressureTail(readTranscriptTail(payload.transcript_path))) return ""; + if (bumpStopCounter(payload.cwd, state) === "release") return ""; + return buildGoalIdleBlock(payload.cwd, state, payload.session_id, platform); + } +``` + +Update the stale comments at `hook.ts:1646-1653,1760-1766` so they no longer say unbound goals receive `cxc loop init` or that the Stop counter bootstraps state. Bound empty goalplans remain bound and continue to get the “register workPhases” guidance (`hook.ts:1687`). Do not change the in-flight B/C continuation branch at `:1801-1815`. + +### MODIFY `plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts` + +- Replace `GOAL-IDLE-CONTINUE-01: active goal at IDLE blocks with the arming command` at `:506-529` with `GOAL-IDLE-CONTINUE-01: active goal without bound plan releases without state write`. Keep the active host goal fixture, assert `handleStop(...) === ""`, `existsSync(join(cwd, ".codexclaw")) === false`, and a second call remains silent. This fails before the fix because the first call blocks and writes state. +- Existing win32 and bounded tests at `:535-565` currently use unbound state. Bind a real plan/slug before calling Stop; retain their platform and cap assertions. Otherwise they would contradict the new contract. +- Keep the bound-plan case at `:567-587` and the bound-empty case at `:589-600`. Add `GOAL-IDLE-CONTINUE-01: stale slug releases without counter write`: write state with a nonexistent slug and `stopBlockTotal: 7`, call Stop, assert empty output and unchanged total. + +## Activation and bypass record + +Exercise no state, unbound state, stale slug, bound populated plan, bound empty plan, active/inactive goal, win32 recipe, and in-flight phase. Tier: Stop-hook continuation control. Executing surface: `handleStop`. Known bypass: other hooks or host goal policies can continue a turn independently. Residual risk: unreadable plan releases because it cannot establish a binding. Wording: “PABCD's IDLE block requires a resolvable bound plan.” Final enforcement layer: Stop handler. Out of scope: changing host-goal completion semantics or `GOAL-COMPLETE-GATE-01`. diff --git a/devlog/_plan/260927_issue_train/014_issue254_turn_budget.md b/devlog/_plan/260927_issue_train/014_issue254_turn_budget.md new file mode 100644 index 00000000..82f1d60d --- /dev/null +++ b/devlog/_plan/260927_issue_train/014_issue254_turn_budget.md @@ -0,0 +1,69 @@ +# #254 — Reset the absolute Stop cap per genuine user turn + +Keep `MAX_STOP_BLOCKS_TOTAL = 24`, but count within one real user turn. A new `UserPromptSubmit` `turn_id` resets the total once; Stop-hook continuations do not reset it. On the 25th attempted block, release with one nonblocking `systemMessage` explaining the cap. + +Current anchors: `plugins/codexclaw/components/pabcd-state/src/state.ts:136`, `plugins/codexclaw/components/pabcd-state/src/hook.ts:624`, `plugins/codexclaw/components/pabcd-state/src/hook.ts:1459`, `plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts:1036`. + +## Runtime proof + +Codex Stop converts `decision:block` to continuation fragments (`/tmp/cxc-perm/codex-src/codex-rs/hooks/src/events/stop.rs:319-391`). The core records those fragments as a response item and continues the same turn (`/tmp/cxc-perm/codex-src/codex-rs/core/src/session/turn.rs:666-683`). UserPromptSubmit is invoked for `TurnInput::UserInput`; a `ResponseItem` receives no such hook (`/tmp/cxc-perm/codex-src/codex-rs/core/src/hook_runtime.rs:682-707`). Thus resetting on a changed UserPromptSubmit `turn_id` excludes Stop continuations in this runtime. The persisted turn ID still protects against duplicate prompt events. If a future native runtime changes this routing, a continuation-origin flag must be added before changing the budget rule. + +## File change map + +### MODIFY `plugins/codexclaw/components/pabcd-state/src/state.ts` + +Add `stopBlockTurnId: string | null` next to `stopBlockTotal` in `State` (`:135-137`), default it to `null` beside `stopBlockTotal: 0` (`:299-301`), and reconstruct a nonempty string or `null` beside `:551-554`. `writeState` already serializes the whole state (`:607-615`); no new writer needed. Old files reconstruct `null`, and a missing `turn_id` does not reset an existing budget. + +```diff + stopBlockTotal: number; ++/** Last genuine UserPromptSubmit turn whose total was reset. */ ++stopBlockTurnId: string | null; + + stopBlockTotal: 0, ++stopBlockTurnId: null, + + stopBlockTotal: + typeof parsed.stopBlockTotal === "number" && Number.isFinite(parsed.stopBlockTotal) && parsed.stopBlockTotal >= 0 + ? Math.floor(parsed.stopBlockTotal) : 0, ++stopBlockTurnId: typeof parsed.stopBlockTurnId === "string" && parsed.stopBlockTurnId.length > 0 ++ ? parsed.stopBlockTurnId : null, +``` + +The numeric branch above is the existing inline validation at `state.ts:551-554`. + +### MODIFY `plugins/codexclaw/components/pabcd-state/src/hook.ts` + +In `handleUserPromptSubmit` at `:629-653`, read state before branch-specific writes. After the existing memory-write marker step (which may update state), re-read state, then reset only when `turn !== ""`, the state file already exists, and `state.stopBlockTurnId !== turn`. Persist `{ ...state, stopBlockTotal: 0, stopBlockTurnId: turn }`, then use this fresh snapshot in the rest of the handler. Do the reset before the `injectedTurns` dedupe, but the persisted turn ID makes a duplicate event a no-op. Do not clear `stopBlockCount`, `stopMetricCursor`, or work-phase progress. For a fresh cwd without a session file, honor #255: no reset write; first verified mutation creates default state, and a later genuine prompt stamps the next turn. + +```ts +let state = readState(payload.cwd, payload.session_id); +if (turn && sessionStateFileExists(payload.cwd, payload.session_id) && state.stopBlockTurnId !== turn) { + state = { ...state, stopBlockTotal: 0, stopBlockTurnId: turn }; + writeState(payload.cwd, state); +} +if (turn && state.injectedTurns.includes(turn)) return ""; +``` + +Use `sessionStateFileExists` added in `011`; do not infer existence from `readState`, which returns a default for absent files (`state.ts:486-603`). Place the reset after `hook.ts:643-651` memory marker so a later spread cannot overwrite either field. `bumpStopCounter` at `hook.ts:1459-1479` continues to increment `stopBlockTotal` for each Stop and release when `nextTotal > MAX_STOP_BLOCKS_TOTAL`. Change its return to distinguish `"phase-cap"` and `"total-cap"`, so only the absolute-cap release emits a message. Every caller at `hook.ts:1791,1810` must handle either release code. + +```diff +-if (nextCount > MAX_STOP_BLOCKS || nextTotal > MAX_STOP_BLOCKS_TOTAL) { ++if (nextCount > MAX_STOP_BLOCKS || nextTotal > MAX_STOP_BLOCKS_TOTAL) { + writeState(cwd, { ...state, ...carry, stopBlockPhase: null, stopBlockWorkPhaseId: null, stopBlockCount: 0 }); +- return "release"; ++ return nextTotal > MAX_STOP_BLOCKS_TOTAL ? "total-cap" : "phase-cap"; + } +``` + +For `total-cap`, return `JSON.stringify({ systemMessage: "CodexClaw Stop continuation cap (24) reached for this user turn; releasing." })` plus newline. Do not include `decision:block`, `continue:false`, or `stopReason`. The universal Stop output accepts `systemMessage` (`/tmp/cxc-perm/codex-src/codex-rs/hooks/src/schema.rs:90-99,455-464`); the runtime records it as a Warning without blocking (`/tmp/cxc-perm/codex-src/codex-rs/hooks/src/events/stop.rs:277-293`). A phase-cap returns `""` as before. + +### MODIFY tests + +- `plugins/codexclaw/components/pabcd-state/test/state.test.ts`: extend the exact default object at `:34-62` with `stopBlockTurnId: null`; add `stopBlockTurnId round trips and malformed value becomes null`. This fails before the field exists. +- `plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts`: update `050 S10/S14: the absolute cap holds against forged progress` at `:1036-1053` to parse the 25th output as a nonblocking `systemMessage`, assert no `decision`, assert `stopBlockTotal === 25`, and ensure earlier 24 blocks remain bounded even with improving metrics. +- Add `absolute Stop cap resets once on new real UserPromptSubmit turn`: seed an existing session with total 24 and turn `old`; send prompt with `turn_id: new`; assert total 0 and `stopBlockTurnId === "new"`; repeat same prompt and assert total stays after a Stop; another new turn resets again. This fails today because total never resets. +- Add `Stop continuation never invokes reset path`: call Stop repeatedly with the same turn but no new UserPromptSubmit, including progress records, assert 25th releases. This is the local behavioral approximation of the native routing proof above. + +## Activation and bypass record + +Exercise absent turn ID, duplicate turn ID, new turn ID, old-schema state, missing state, same-turn continuation, per-phase cap, and absolute cap. Tier: PABCD Stop-hook limit. Executing surface: UserPromptSubmit bookkeeping plus Stop decision. Known bypass: direct state edits or a native implementation that routes internal response items as user input; residual risk: future Codex runtime routing drift. Wording: “24 blocks per observed genuine user turn on the verified runtime.” Final enforcement layer: Stop `bumpStopCounter`. Out of scope: changing the 24/3 constants, host model retry budgets, or native Codex code. diff --git a/devlog/_plan/260927_issue_train/015_issue251_worker_gate.md b/devlog/_plan/260927_issue_train/015_issue251_worker_gate.md new file mode 100644 index 00000000..632b98de --- /dev/null +++ b/devlog/_plan/260927_issue_train/015_issue251_worker_gate.md @@ -0,0 +1,37 @@ +# #251 — Gate legacy worker only during an armed PABCD build/check cycle + +Registered `executor` always needs an evidence receipt. A built-in `worker` needs one only when its parent session is actively orchestrating phase B or C. An ordinary worker releases at its first SubagentStop, without attempt files or tombstones. + +Current anchors: `plugins/codexclaw/components/pabcd-state/src/subagent-evidence.ts:471`, `plugins/codexclaw/components/pabcd-state/src/state.ts:486`, `plugins/codexclaw/components/pabcd-state/test/subagent-evidence.test.ts:57`, `plugins/codexclaw/skills/pabcd/references/delegation.md:12`, `plugins/codexclaw/components/subagent-config/src/spawn-wrapper.ts:27`. + +## File change map + +### MODIFY `plugins/codexclaw/components/pabcd-state/src/subagent-evidence.ts` + +`GATED_AGENT_TYPES` currently includes both roles (`:57-64`) and `runSubagentStopGate` enters receipt logic immediately (`:471-510`). Keep the matcher set so registered hooks still reach this function; add the worker predicate before `extractReceiptPath`, `readAttempts`, or any state/evidence write. + +```ts +if (!GATED_AGENT_TYPES.has(payload.agent_type)) return ""; +if (payload.agent_type === "worker") { + const { state, unreadable } = readStateStrict(payload.cwd, payload.session_id); + if (unreadable || !state.orchestrationActive || (state.phase !== "B" && state.phase !== "C")) return ""; +} +``` + +`readStateStrict` is the existing non-throwing strict reader (`state.ts:486-603`); `orchestrationActive` is reconstructed false at IDLE (`state.ts:529`). The chosen predicate is **both** active orchestration and phase B/C. Phase P/A reviewers and ordinary worker delegations stay outside this receipt gate; only actual build/check workers need the legacy fallback. `executor` bypasses the predicate and retains the existing receipt, retry, tombstone, and parent-completion consequences. An unreadable worker state releases because the parent cycle cannot be proved armed; executor still follows the current fail-safe/tombstone behavior. Keep all existing receipt root validation for gated paths. + +### MODIFY `plugins/codexclaw/components/pabcd-state/test/subagent-evidence.test.ts` + +At `:57-64`, replace the unarmed default-worker expectation with `worker outside armed PABCD releases without attempts or tombstone`: assert output `""`, `readAttempts(...) === 0`, no `.codexclaw/evidence-attempts` path, and `readState(...).unverifiedSubagents` empty. It fails before the fix because first Stop blocks. For the existing receipt/tombstone tests that use the test helper's default worker, either seed `{ phase:"B", orchestrationActive:true }` in each fixture or change only the helper default to `executor`; preserve explicit worker coverage. Add `worker in armed B and C blocks without receipt`, `worker at P/A/IDLE or orchestrationActive false releases`, `executor without active cycle still blocks`, and `worker with unreadable state releases without write`. Assert no attempt/tombstone writes on every release path. The `GATED_AGENT_TYPES` set assertion can remain (`test/:57` and later role checks): it describes hook routing, not unconditional gate application. + +### MODIFY `plugins/codexclaw/skills/pabcd/references/delegation.md` + +At `:8-19`, add one sentence after the registered-executor fallback line: “The registered executor is evidence-gated on every SubagentStop; the built-in worker fallback is evidence-gated only while the parent has an active PABCD B/C cycle. Outside that cycle the worker releases without a receipt.” + +### MODIFY `plugins/codexclaw/components/subagent-config/src/spawn-wrapper.ts` + +Update the comment at `:8-12` to mention this distinction. Do not change `ROLE_AGENT_TYPE` (`:26-32`): unregistered executor still maps to native `worker`, and the runtime gate uses parent session phase. If the canonical doctrine in `structure/20_pabcd_dispatch_doctrine.md` repeats “all workers always gated,” update that exact sentence in the builder branch after locating it; this is a documentation sync, not a runtime dependency. + +## Activation and bypass record + +Exercise executor with no state, worker with no state, worker P/A/B/C/IDLE, B/C with `orchestrationActive=false`, corrupt state, valid/invalid receipt, repeated Stop, and tombstone terminal behavior. Tier: cooperative SubagentStop enforcement; executing surface: `runSubagentStopGate`; known bypass: a child labeled as an ungated type or a missing/untrusted hook; residual risk: a worker that performs writes outside an armed cycle is no longer receipt-gated. Wording: “legacy worker receipts are required in active B/C cycles,” not “every worker is verified.” Final enforcement layer: SubagentStop runtime gate, followed by parent completion gate for recorded tombstones. Out of scope: changing native agent registration or receipt file format. diff --git a/devlog/_plan/260927_issue_train/016_issue250_trigger_narrowing.md b/devlog/_plan/260927_issue_train/016_issue250_trigger_narrowing.md new file mode 100644 index 00000000..a142aed9 --- /dev/null +++ b/devlog/_plan/260927_issue_train/016_issue250_trigger_narrowing.md @@ -0,0 +1,84 @@ +# #250 — Require an explicit CodexClaw phase or loop request + +Incidental `interview`, Korean build/verify verbs, and ordinary “keep going” language must not inject PABCD context or arm the loop. Explicit line-anchored `orchestrate ` remains authoritative through `parseOrchestrateCommand`; this change narrows only the advisory detectors. + +Current anchors: `plugins/codexclaw/components/pabcd-state/src/hook.ts:238`, `plugins/codexclaw/components/pabcd-state/src/hook.ts:272`, `plugins/codexclaw/components/pabcd-state/src/hook.ts:655`, `plugins/codexclaw/components/pabcd-state/test/hook.test.ts:63`, `plugins/codexclaw/components/pabcd-state/test/hook.test.ts:461`. + +## File change map + +### MODIFY `plugins/codexclaw/components/pabcd-state/src/hook.ts` + +`detectTrigger` at `:238-250` currently matches bare `interview`/`인터뷰`, generic `plan this`, `build this`, Korean `구현해`/`검증해`, and `audit this`. Replace with a line-oriented explicit marker grammar. An accepted line must name `cxc-pabcd`, `codexclaw:cxc-pabcd`, a `[$cxc-pabcd](skill://...)` mention, `PABCD로`/`PABCD phase`, or a literal line-start `orchestrate [ipabc]`; for phase selection, it must additionally carry an unambiguous phase token (`interview/I`, `plan/P`, `audit/A`, `build/B`, `check/C`) on that line. A line-start `orchestrate` token remains handled by the existing parser first (`hook.ts:655-664`); the detector may recognize it for pure-function compatibility but must not broaden command parsing. Do not scan the whole prompt with an unanchored regex: quoted issue bodies often contain exact tokens. Add this complete helper before both detectors and replace `detectTrigger` with: + +```ts +function requestLines(prompt: string): string[] { + const result: string[] = []; + let fenced = false; + for (const raw of (prompt ?? "").split(/\r?\n/)) { + const line = raw.trim(); + if (/^```/.test(line)) { fenced = !fenced; continue; } + if (fenced || !line || /^(?:>|[-*] |\d+[.)] )/.test(line)) continue; + result.push(line); + } + return result; +} + +export function detectTrigger(prompt: string): Phase | null { + for (const line of requestLines(prompt)) { + const command = /^orchestrate\s+([ipabc])(?:\s|$)/i.exec(line); + if (command) { + const phase = command[1].toUpperCase(); + if (phase === "I" || phase === "P" || phase === "A" || phase === "B" || phase === "C") return phase; + } + const marker = /(?:\bcxc-pabcd\b|\bcodexclaw:cxc-pabcd\b|\[\$?cxc-pabcd\]\(skill:\/\/[^)]+\)|\bpabcd\s*(?:로|phase\b))/i.test(line); + const requested = /(?:\b(?:use|run|start|enter|apply)\b|(?:시작|진행|적용|돌려|들어가))/i.test(line); + if (!marker || !requested) continue; + if (/\binterview\b|(?:^|\s)인터뷰(?:\s|$)|\bphase\s*i\b/i.test(line)) return "I"; + if (/\bplan\b|\bphase\s*p\b|계획/.test(line)) return "P"; + if (/\baudit\b|\bphase\s*a\b|감사/.test(line)) return "A"; + if (/\bbuild\b|\bphase\s*b\b|구현/.test(line)) return "B"; + if (/\bcheck\b|\bphase\s*c\b|검증/.test(line)) return "C"; + if (/pabcd\s*로|\bcxc-pabcd\b/i.test(line)) return "P"; + } + return null; +} +``` + +The helper excludes quoted, list, and fenced examples. Preserve phase priority only within one explicitly requested line, and update old tests that asserted generic phrasing (`hook.test.ts:63-99`). + +`detectLoopArmRequest` at `:272-290` currently accepts unanchored goalplan/HOTL tokens, repeated Korean phrases, generic “keep going until,” and “끝까지 진행해.” Replace with the same line scanner/quoted-text exclusion. Accept a request line that names `cxc-loop`, `codexclaw:cxc-loop`, a matching skill link, `goalplan`/`골플랜`, `HOTL`, or `PABCD` and has a run/arm/create imperative on that line. Bare marker mentions or general persistence language return false. In particular delete the branches at `:284-288`; do not add a fallback on `continue until done`. Keep `parseOrchestrateCommand` untouched. Full replacement: + +```ts +export function detectLoopArmRequest(prompt: string): boolean { + for (const line of requestLines(prompt)) { + const marker = /\bcxc-?loop\b|\bcodexclaw:cxc-loop\b|\[\$?cxc-loop\]\(skill:\/\/[^)]+\)|\bgoal\s*plan\b|\bgoalplan\b|골플랜|\bhotl\b|\bi?pabcd\b/i; + const action = /\b(?:use|run|start|arm|create|init|repeat|cycle)\b|(?:돌려|돌리|시작|등록|진행|적용|해줘)/i; + if (marker.test(line) && action.test(line)) return true; + } + return false; +} +``` + +In `handleUserPromptSubmit`, the loop-arm branch at `:690-711` persists `loopArmSeen`; incidental prompts must take the silent path and leave it false. Existing explicit command path at `:655-664` is unchanged. A prompt with a literal skill mention and imperative must still inject once per turn. + +### MODIFY `plugins/codexclaw/components/pabcd-state/test/hook.test.ts` + +Replace generic positive expectations at `:63-99,461-494` with explicit marker plus action cases. Also update the C2 lexical-trigger fixtures at `hook.test.ts:182-263`: prompts whose only phase signal is the ordinary Korean verify verb now expect null and silent handler output; where a test is meant to prove the CHECK directive itself, add an explicit `cxc-pabcd` marker and action to its prompt. Add a table test named `issue 250: nine reported prompts stay silent` with the exact issue examples and expected `(trigger, loopArm)`: + +| Label | Prompt | Expected | +| --- | --- | --- | +| order_line | `Keep going until Done means holds. ...` | null, false | +| wn_workflow_name | `... workflow 인터뷰엔진 must keep its name.` | null, false | +| author_ko_build | `이 기능 구현해 두고 결과 보고해` | null, false | +| author_ko_verify | `검증해 보고 알려줘` | null, false | +| author_ko_finish | `끝까지 진행해` | null, false | +| english_mention | `Summarize the interview notes in file X` | null, false | +| neg_thanks | `감사합니다` | null, false | +| neg_for_loop | `fix the for loop bug in parser.ts` | null, false | +| neg_plain | `list the files in out/` | null, false | + +These prompts and the old false positives are recorded in issue #250. Add `issue 250: incidental prompt emits no context and does not arm`: call `handleUserPromptSubmit` for each on a fresh existing default-state fixture, assert output `""`, `loopArmSeen === false`, `injectedTurns` unchanged. The six formerly positive examples fail before the fix. Add `issue 250: explicit skill request arms once` with `Use [$cxc-pabcd](skill:///Users/jun/.codex/plugins/cache/codexclaw/codexclaw/0.2.39+codex.20260924082502/skills/pabcd/SKILL.md) to start Plan phase` and `Run cxc-loop for this task`; assert the first emits context and the duplicate `turn_id` is silent. Add quoted/fenced mention negatives and line-start `orchestrate i` positive. + +## Activation and bypass record + +Exercise each of nine issue prompts, explicit English/Korean markers, skill link, duplicate turn, quoted line, fenced code, mixed incidental and explicit lines, and explicit orchestrate grammar. Tier: advisory prompt detector; executing surface: `handleUserPromptSubmit`; known bypass: other installed skills or direct human/CLI orchestration can still enter PABCD; residual risk: an unusual natural-language request lacking a marker no longer gets a hint. Wording: “automatic hints require an explicit CodexClaw request”; no claim that all phase entry is disabled. Final enforcement layer: detector plus existing parser-first dispatch. Out of scope: changing explicit CLI/chat command grammar, goal policy, or unrelated search-request detection. diff --git a/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md b/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md new file mode 100644 index 00000000..7b0f9ff2 --- /dev/null +++ b/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md @@ -0,0 +1,285 @@ +# wp3 — Agent-created thread permission hook and advisory + +Codex Desktop can start a `create_thread` child with approval prompts even when the user's top-level Codex configuration says `never` and `danger-full-access`. This phase offers a user-global, default-off PermissionRequest auto-allow for that exact provenance and a separate SessionStart warning that works without the opt-in. It does not claim to change the child's sandbox or network access. The symptom also occurs for projectless children, so no worktree predicate belongs in the eligibility test. + +## Phase contract + +- Class: C4, permission boundary. Binding decisions: AD-1 through AD-5 and AD-7 in `devlog/_plan/260927_issue_train/002_architect_consultation.md:9-15`. +- Dependency: wp2's SessionStart no-write behavior and PABCD switch. The module itself only reads files and returns JSON and does not call `handleSessionStart`; the shared CLI path still records a hook observation under `CODEX_HOME` (plugins/codexclaw/scripts/hook-observation.mjs:17,70), never under the cwd. Both verbs stay active when `CODEXCLAW_PABCD=off` or `pabcd.enabled=false`, because they are not PABCD policy; 012's switch must not list them. +- Success: with the explicit global opt-in, an agent-created root thread whose hook says `permission_mode: "default"` and whose user Codex config shows explicit top-level evidence `approval_policy = "never"` and `sandbox_mode = "danger-full-access"` (no profile) emits the exact allow object for Bash, write_stdin, apply_patch, request_permissions, and tool names that follow the `mcp____` naming convention. This is config evidence of user intent, not proof of the thread's effective permission; an exact guarantee needs Codex to expose the resolved policy and sandbox in PermissionRequest input. Every missing, mismatched, corrupt or unknown input emits zero stdout bytes and exits 0. The advisory is independent of opt-in. +- Runtime proof boundary: hook input has `session_id`, `transcript_path`, `permission_mode`, `tool_name`, and optional `agent_id`/`agent_type` at `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/schema.rs:298-318`; SessionStart input has the first three at `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/schema.rs:496-510`. `SessionMeta` stores `id` and `thread_source` at `/tmp/cxc-perm/codex-src/codex-rs/protocol/src/protocol.rs:3128-3154`, and `ThreadSource::Feature` serializes its feature string at `/tmp/cxc-perm/codex-src/codex-rs/protocol/src/protocol.rs:2841-2857`. The specific `agent_created_thread` value is the observed rollout fixture from this issue train, not a universal enum variant. + +## File change map + +All paths below are repository-relative. Anchors were rechecked at `codex/issue-train-0927` / `958441a9` on 2026-09-27. Add no package dependency. + +### 1. NEW `plugins/codexclaw/components/pabcd-state/src/agent-thread-permissions.ts` + +The module owns both event handlers and all eligibility reads. `CODEXCLAW_HOME` uses the existing user-global default at `plugins/codexclaw/components/subagent-config/src/store.ts:128-133`; unlike `subagents.json`, the new opt-in lives in `config.json`. This permission reader rejects relative home overrides so a cwd-local path cannot become an accidental opt-in. Do not read repo-local `codexclaw.json` (`plugins/codexclaw/components/pabcd-state/src/interview-policy.ts:26-27`) or `.codexclaw/*`. The existing `cxc config set` writes whitelisted Codex `config.toml` keys (`plugins/codexclaw/components/config-guard/src/cli.ts:27`, `plugins/codexclaw/components/config-guard/src/config-set.ts:62-74`), while `cxc config interview` writes project-local policy (`plugins/codexclaw/components/pabcd-state/src/cli.ts:209-245`). Do **not** add a CLI toggle: adding a second global JSON writer to those routes enlarges the permission surface for one opt-in. Document manual editing of `$CODEXCLAW_HOME/config.json` (default `~/.codexclaw/config.json`): + +```json +{"permissions":{"agentCreatedThreadAutoAllow":true}} +``` + +The value must be the JSON boolean `true`; missing file/key, malformed JSON, arrays, and string `"true"` are off. Preserve other keys if editing an existing file. Add the following module verbatim; its 64 KiB first-record cap and 1 MiB config cap are named conservative bounds. It recognizes only the two required top-level TOML string assignments and refuses duplicates, an active `profile` key, or malformed section boundaries. This is deliberately narrower than a full TOML parser: the existing readers in `pabcd-state/src/review-round-cli.ts:36-50` and `config-guard/src/toml-edit.ts:51-69` are table/line readers, not a general parser. A syntax it cannot establish yields no allow. + +```ts +import { closeSync, openSync, readFileSync, readSync, statSync } from "node:fs"; +import { homedir } from "node:os"; +import { isAbsolute, join } from "node:path"; + +const MAX_META_LINE_BYTES = 64 * 1024; +const MAX_CONFIG_BYTES = 1024 * 1024; +const ALLOW = '{"hookSpecificOutput":{"hookEventName":"PermissionRequest","decision":{"behavior":"allow"}}}'; +const MODEL_ADVICE = "This Codex Desktop agent-created thread may show approval prompts even though the user Codex config requests full access. Request escalation explicitly for network or git operations when needed; do not assume this hook changes sandbox or network access."; +const USER_ADVICE = "This agent-created thread started in the default approval mode despite your full-access Codex config, so approval prompts may appear. You can switch this thread to Full Access in the composer or enable permissions.agentCreatedThreadAutoAllow in your user-global Codexclaw config."; + +type JsonObject = Record; + +function object(value: unknown): JsonObject | null { + return value !== null && typeof value === "object" && !Array.isArray(value) + ? value as JsonObject : null; +} + +function readFirstRecord(path: string): JsonObject | null { + if (!isAbsolute(path)) return null; + const fd = openSync(path, "r"); + try { + const buffer = Buffer.alloc(MAX_META_LINE_BYTES + 1); + let used = 0; + while (used < buffer.length) { + const count = readSync(fd, buffer, used, buffer.length - used, used); + if (count === 0) break; + used += count; + const end = buffer.subarray(0, used).indexOf(10); + if (end >= 0) return end > MAX_META_LINE_BYTES ? null : + object(JSON.parse(buffer.subarray(0, end).toString("utf8"))); + } + if (used === 0 || used > MAX_META_LINE_BYTES) return null; + return object(JSON.parse(buffer.subarray(0, used).toString("utf8"))); + } finally { + closeSync(fd); + } +} + +function agentCreatedRoot(input: JsonObject): boolean { + if (input.permission_mode !== "default" || + typeof input.session_id !== "string" || input.session_id.length === 0 || + typeof input.transcript_path !== "string" || + Object.hasOwn(input, "agent_id") || Object.hasOwn(input, "agent_type")) return false; + const record = readFirstRecord(input.transcript_path); + const payload = object(record?.payload); + return record?.type === "session_meta" && payload?.id === input.session_id && + payload.thread_source === "agent_created_thread" && payload.forked_from_id == null; +} + +function boundedText(path: string): string | null { + const stat = statSync(path); + if (!stat.isFile() || stat.size > MAX_CONFIG_BYTES) return null; + return readFileSync(path, "utf8"); +} + +function globalOptIn(env: NodeJS.ProcessEnv): boolean { + const override = env.CODEXCLAW_HOME?.trim(); + if (override && !isAbsolute(override)) return false; + const home = override || join(homedir(), ".codexclaw"); + const config = object(JSON.parse(boundedText(join(home, "config.json")) ?? "null")); + const permissions = object(config?.permissions); + return permissions?.agentCreatedThreadAutoAllow === true; +} + +function codexConfigFullAccess(env: NodeJS.ProcessEnv): boolean { + const override = env.CODEX_HOME?.trim(); + if (override && !isAbsolute(override)) return false; + const home = override || join(homedir(), ".codex"); + const content = boundedText(join(home, "config.toml")); + if (content === null) return false; + const seen = new Map(); + for (const raw of content.replace(/^\uFEFF/, "").split(/\r?\n/)) { + const line = raw.trim(); + if (line === "" || line.startsWith("#")) continue; + if (line.startsWith("[")) { + // An invalid table header leaves the ownership of following keys unknown. + if (!/^\[\[?[^\]\r\n]+\]\]?\s*(?:#.*)?$/.test(line)) return false; + break; + } + if (/^(?:profile|"profile"|'profile')\s*=/.test(line)) return false; + const key = /^(approval_policy|sandbox_mode)\s*=/.exec(line)?.[1]; + if (!key) { + if (!/^[A-Za-z0-9_-]+\s*=\s*\S/.test(line)) return false; + continue; + } + if (seen.has(key)) return false; + const match = new RegExp(`^${key}\\s*=\\s*"([^"\\r\\n]*)"\\s*(?:#.*)?$`).exec(line); + if (!match) return false; + seen.set(key, match[1]); + } + return seen.get("approval_policy") === "never" && + seen.get("sandbox_mode") === "danger-full-access"; +} + +function coveredTool(name: unknown): boolean { + return typeof name === "string" && + (["Bash", "write_stdin", "apply_patch", "request_permissions"].includes(name) || + /^mcp__/.test(name)); +} + +function parseHook(raw: string, event: string): JsonObject | null { + const input = object(JSON.parse(raw)); + return input?.hook_event_name === event ? input : null; +} + +export function handleAgentThreadPermissionRequest( + raw: string, env: NodeJS.ProcessEnv = process.env, +): string { + try { + const input = parseHook(raw, "PermissionRequest"); + return input && coveredTool(input.tool_name) && globalOptIn(env) && + agentCreatedRoot(input) && codexConfigFullAccess(env) ? ALLOW : ""; + } catch { + return ""; + } +} + +export function handleAgentThreadSessionStartAdvisory( + raw: string, env: NodeJS.ProcessEnv = process.env, +): string { + try { + const input = parseHook(raw, "SessionStart"); + if (!input || !agentCreatedRoot(input) || !codexConfigFullAccess(env)) return ""; + return `${JSON.stringify({ + systemMessage: USER_ADVICE, + hookSpecificOutput: { hookEventName: "SessionStart", additionalContext: MODEL_ADVICE }, + })}\n`; + } catch { + return ""; + } +} +``` + +### 2. MODIFY `plugins/codexclaw/components/pabcd-state/src/cli.ts` + +Current `readStdin` and overflow handling are at lines 68-109 and 333-340; `recordHookInvocation` is line 340; the root-only early exit is lines 389-393. Insert the two hook verbs immediately after `const raw = stdin.raw` and before the recorder and early exit. Handle oversized input for these verbs before the existing branch's exit-1 behavior; AD-1 requires fail-open exit 0. The module rejects agent fields itself, so do not rely on the old early exit as the permission boundary. + +```diff + const stdin = readStdin(); + if (stdin.overflow) { ++ if (event === "permission-request" || event === "session-start-permission-advisory") { ++ process.exit(0); ++ } + const denied = oversizedHookOutput(event); +@@ + const raw = stdin.raw; ++ if (event === "permission-request" || event === "session-start-permission-advisory") { ++ try { ++ recordHookInvocation(raw, "pabcd-state", event, import.meta.url); ++ const { handleAgentThreadPermissionRequest, handleAgentThreadSessionStartAdvisory } = ++ await import("./agent-thread-permissions.ts"); ++ const result = event === "permission-request" ++ ? handleAgentThreadPermissionRequest(raw) ++ : handleAgentThreadSessionStartAdvisory(raw); ++ if (result) process.stdout.write(result); ++ } catch { ++ // Fail open: no decision/advisory, exit 0. ++ } ++ process.exit(0); ++ } + recordHookInvocation(raw, "pabcd-state", event, import.meta.url); +``` + +No other event flow changes. In particular, the existing generic `session-start` handler remains side-effect-only at `cli.ts:406-410`; the new advisory does not mutate `.codexclaw` session state. + +### 3. NEW `plugins/codexclaw/hooks/permission-request-allowing-agent-thread.json` + +Use matcher `"*"` because the runtime may vary tool names; the module self-filters. The runtime accepts `*` as match-all at `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/events/common.rs:166,192-194`. A synchronous command is essential: an async hook cannot apply the allow decision. The output wire's exact `hookSpecificOutput` and `decision.behavior` schema is `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/schema.rs:161-165,189-225`; the parser accepts it at `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/engine/output_parser.rs:184-204`. + +```json +{ + "hooks": { + "PermissionRequest": [{ + "matcher": "*", + "hooks": [{ + "type": "command", + "command": "node \"${PLUGIN_ROOT}/components/pabcd-state/dist/cli.js\" hook permission-request", + "timeout": 10, + "statusMessage": "(codexclaw) Checking agent-created thread permission" + }] + }] + } +} +``` + +### 4. NEW `plugins/codexclaw/hooks/session-start-advising-agent-thread-permissions.json` + +Follow `plugins/codexclaw/hooks/session-start-bootstrapping-pabcd-state.json:1-16`, but use a separate verb. `systemMessage` is in the universal output at `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/schema.rs:87-99`; SessionStart `additionalContext` is at `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/schema.rs:384-403` and read by `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/engine/output_parser.rs:93-99`. + +```json +{ + "hooks": { + "SessionStart": [{ + "hooks": [{ + "type": "command", + "command": "node \"${PLUGIN_ROOT}/components/pabcd-state/dist/cli.js\" hook session-start-permission-advisory", + "timeout": 10, + "statusMessage": "(codexclaw) Advising on agent-created thread permissions" + }] + }] + } +} +``` + +### 5. MODIFY `plugins/codexclaw/.codex-plugin/plugin.json` + +The manifest currently lists 29 hooks at lines 22-52. Add both paths once, beside the other SessionStart entries and before the PreToolUse entries. Their ordering among different event types does not determine policy. + +```diff + "./hooks/session-start-announcing-map-affordance.json", ++ "./hooks/session-start-advising-agent-thread-permissions.json", +@@ + "./hooks/pre-tool-use-guarding-goal-budget.json", ++ "./hooks/permission-request-allowing-agent-thread.json", +``` + +### 6. MODIFY `plugins/codexclaw/inventory.json` and `README.md`, `README.ko.md`, `README.zh.md` + +Run the full `npm test` first and read its measured total, then `node plugins/codexclaw/scripts/inventory.mjs --write --tests ` after the manifest and hook files exist, then `inventory.mjs --check --tests ` and `npm run gate`. The generator derives hook identities from manifest entries at `plugins/codexclaw/scripts/inventory.mjs:77-92`, compares manifest and hook-file sets at lines 163-197, and updates the three README hook badges at lines 305-321. Expected count: 29 -> 31 (`README.md:18`, `README.ko.md:18`, `README.zh.md:18`). Then manually change the stale literal `24 active hooks` to `31 active hooks` at `README.md:189`, `README.ko.md:179`, and `README.zh.md:178`, plus `approve the 24 hooks` to `approve the 31 hooks` at `README.md:71`. The generator does not update those prose counts. + +### 7. NEW `plugins/codexclaw/components/pabcd-state/test/agent-thread-permissions.test.ts` + +Use `node:test` and `node:assert/strict`, as in adjacent component tests. Fixture helper makes temporary `CODEX_HOME/config.toml`, `CODEXCLAW_HOME/config.json`, and a rollout JSONL file whose first line is `{"type":"session_meta","payload":{"id":"fixture-id","thread_source":"agent_created_thread"}}\n`; pass those env vars to the exported handlers, with absolute `transcript_path`. Use `mkdtempSync(join(tmpdir(), ...))` and `t.after(() => rmSync(dir,{recursive:true,force:true}))`. The positive payload is `hook_event_name:"PermissionRequest", session_id:"fixture-id", permission_mode:"default", tool_name:"Bash"`; global JSON opt-in is boolean true; TOML has the two top-level required assignments. Test names and assertions: + +| Named test | Exact assertion and reason it fails before this phase | +|---|---| +| `allows opted-in agent-created root thread for covered tools` | For `Bash`, `write_stdin`, `apply_patch`, `request_permissions`, `mcp__codex_app__create_worktree`, assert `strictEqual(result, '{"hookSpecificOutput":{"hookEventName":"PermissionRequest","decision":{"behavior":"allow"}}}')`. No handler exists before this phase. | +| `default-off global setting leaves approval to Codex` | For missing `config.json`, missing `permissions`, malformed JSON, `permissions` array, `false`, and string `"true"`, assert empty output. The new gate must not silently grant approval. | +| `rejects missing corrupt and mismatched rollout identity` | Subcases: missing/empty `session_id`, null/missing transcript path, nonexistent path, empty file, malformed first JSONL line, first line not `session_meta`, missing payload, missing id, wrong id, missing/wrong `thread_source`, first line >64 KiB, and a valid second line after a wrong first line; assert empty output each. Without first-record verification an arbitrary thread could be allowed. | +| `rejects non-default permission and subagent payloads` | `permission_mode` missing/`full-access` and present `agent_id` or `agent_type` (including empty string or null) each produce empty output. An inherited child must never use this exception. | +| `rejects forked and user thread sources` | First-line `thread_source:"user"`, `"subagent"`, and `thread_source:"agent_created_thread"` with non-null `forked_from_id` all produce empty output, even with opt-in. Fork provenance remains outside this exception until measured. | +| `rejects unknown or conflicting Codex config` | Missing TOML, malformed/torn required assignment, malformed top-level line, duplicate required key, missing either key, `on-request`, `workspace-write`, top-level `profile = "team"`, and quoted top-level `"profile" = "team"` each produce empty output. The global opt-in cannot override unknown effective policy. | +| `ignores project-local opt-in and noncovered tool` | Write a project `codexclaw.json`/`.codexclaw/config.json` true but omit global opt-in: empty; set relative `CODEXCLAW_HOME` or `CODEX_HOME`: empty; with absolute global paths and opt-in true, `Edit`, `mcp_tool`, `functions.exec`, and empty tool name: empty. The hook matcher is broad but the handler is not. | +| `emits exact allow JSON and otherwise no stdout` | Parse the positive output and assert only `hookSpecificOutput.hookEventName`/`decision.behavior` keys; assert the output bytes equal the literal JSON above with no trailing LF. Test malformed hook JSON, missing/wrong `hook_event_name`, and thrown read errors yield `""`; no `deny`, `continue:false`, or stderr path exists. | +| `advises agent-created default thread without opt-in` | Remove global opt-in, call SessionStart handler, parse output, assert nonempty `systemMessage` mentioning composer Full Access and opt-in, and `hookSpecificOutput = {hookEventName:"SessionStart",additionalContext:}` with `network` and `git` in the context. This proves AD-5 is independent of AD-2. | +| `advisory is silent outside degraded agent-created context` | `approval_policy:on-request`, `sandbox_mode:workspace-write`, profile present, `permission_mode` other than default, `thread_source:user`, bad first line, and present agent field each yield `""`. Do not tell ordinary threads they degraded. | +| `CLI hooks fail open before root-only subagent exit` | Spawn the built CLI for both verbs with the fixture stdin; assert exit 0 and exact positive outputs. Spawn malformed and >4 MiB inputs; assert exit 0 and empty stdout. This pins placement around `cli.ts:333-340,389-393`; before the change these verb outputs are absent or oversized input exits 1. | + +The first two rows alone are not sufficient: every conditional branch in `agentCreatedRoot`, `globalOptIn`, `codexConfigFullAccess`, `coveredTool`, `parseHook`, the 64 KiB cap, CLI overflow, and the advisory has an activating case above. Use table-driven subtests to keep the file compact. + +### 8. MODIFY `plugins/codexclaw/components/cxc-ops/test/hook-trust.test.ts` + +The existing real-hook golden tests at lines 53-76 and `EVENT_LABELS.PermissionRequest` at `components/cxc-ops/src/hook-trust.ts:18-29` establish the pattern. Add a test named `new agent thread hooks have stable trust identities and both require trust` after line 76. Parse each new hook JSON, pass its event, matcher and handler to `identityHash`, assert a `sha256:` hash, assert the PermissionRequest matcher is exactly `*`, and assert `listHookEntries(PLUGIN_ROOT, "codexclaw@local")` contains exactly one `:permission_request:0:0` key for the new file and exactly one `:session_start:0:0` key for its advisory. With a temporary `CODEX_HOME` containing no trust entries, call `diagnoseHookTrust` and assert both keys are `untrusted`; after writing one matching `trustSection` and one stale hash, assert trusted versus drifted statuses. This would fail before the manifest entries existed, and guards the fact that added hooks are inert until Codex trusts them. + +## Runtime decision and bypass record + +The PermissionRequest surface can suppress a user approval prompt only after all predicates pass. An empty stdout and exit 0 is no decision, so Codex keeps its normal prompt path; this is supported by `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/events/permission_request.rs:205-211`. The exact allow is parsed at `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/engine/output_parser.rs:184-204`. Never emit a denial, exit 2, `updatedInput`, `updatedPermissions`, or `interrupt`: the latter fields are unsupported at `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/schema.rs:199-217`, and exit 2 can deny at `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/events/permission_request.rs:249-263`. Another PermissionRequest hook's deny wins over this allow at `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/events/permission_request.rs:149-169,301-327`. + +Residual risks to record in the implementation PR: top-level `config.toml` can differ from runtime overrides; `permission_mode:"default"` does not prove the active sandbox; an allow result does not widen filesystem or network permissions; the `^mcp__` name check is a naming heuristic; another hook may deny; a new/modified hook declaration changes its trust hash and needs re-approval. The advisory therefore says only that approval mode may have degraded. This workaround does not assert upstream #33282, #40793, or #41167 is fixed. + +## Verification and activation + +Run `npm run build` first (the CLI subprocess subtests execute `dist/cli.js`); then `node plugins/codexclaw/scripts/test.mjs "plugins/codexclaw/components/pabcd-state/test/agent-thread-permissions.test.ts" "plugins/codexclaw/components/cxc-ops/test/hook-trust.test.ts"`; then full `npm test`, `inventory.mjs --write/--check --tests ` and `npm run gate`. Inspect command exit codes and actual test counts. + +Cross-phase integration test (required, in `agent-thread-permissions.test.ts`): run the built CLI for both `permission-request` and `session-start-permission-advisory` with `CODEXCLAW_PABCD=off` and a fresh temporary cwd; assert the positive allow bytes and the advisory JSON are still produced and that `/.codexclaw` does not exist afterwards. Use a scratch `CODEX_HOME` and `CODEXCLAW_HOME`; verify positive PermissionRequest stdout bytes and advisory JSON, then remove the opt-in and verify empty stdout. New manifest hooks require explicit Codex hook trust on an installed plugin before live activation; the source-level tests do not prove desktop permission behavior or sandbox/network access. + +## Out of scope + +No upstream Codex patch, runtime permission-profile change, automatic global opt-in, repo-local permission setting, forked-thread allowance, trust-state forging, or network/sandbox widening. `021_wp3_dispatch_guidance.md` owns the separate dispatch and #265 checkpoint text. diff --git a/devlog/_plan/260927_issue_train/021_wp3_dispatch_guidance.md b/devlog/_plan/260927_issue_train/021_wp3_dispatch_guidance.md new file mode 100644 index 00000000..af489c1e --- /dev/null +++ b/devlog/_plan/260927_issue_train/021_wp3_dispatch_guidance.md @@ -0,0 +1,245 @@ +# wp3 — Dispatch guidance and optional worker checkpoint + +A Codex Desktop thread created through `create_thread` may begin with approval prompts even when its full-access coordinator and user Codex config do not. For bounded work that needs an isolated checkout but not its own goal or PABCD state, the coordinator can create a managed worktree and give its absolute path to a subagent, which passes that path as the shell workdir on every command. Keep separate threads for lanes that must own their own goal, PABCD cycle, or user-visible task. This is a guidance change, not a new dispatcher or permission guarantee. + +## Phase contract + +- Binding decisions: AD-6 in `devlog/_plan/260927_issue_train/002_architect_consultation.md:14`; #265 is an optional worker progress checkpoint. +- Scope: edit only the two skill references below. The permission hook and advisory are specified in `020_wp3_agent_thread_permissions.md`. +- Completion: the two references give the same choice rule, state that the subagent's native cwd still inherits the coordinator's, require disjoint worktrees and an explicit per-command shell workdir, forbid concurrent branch operations in one checkout, and give a replacement worker enough on-disk state to resume an interrupted write packet. +- Existing evidence: `dispatch-surfaces.md:53-58` records inherited cwd; `delegation.md:227-236` records the shared-tree hazard. The permission issue also affects projectless `create_thread` targets; do not describe it as specific to worktree threads. `020` cites the Codex hook/rollout ownership source. + +## File change map + +Anchors are current at `codex/issue-train-0927` / `958441a9` on 2026-09-27. Each diff below is an exact text replacement against the current file. No source or test file changes in this unit. + +### 1. MODIFY `plugins/codexclaw/skills/pabcd/references/dispatch-surfaces.md` + +At lines 21-29, distinguish an independent task lane from a bounded checkout worker: + +```diff + Isolation comes from the environment, not from being a task. A `local` thread is + an independent owner sharing one checkout; a `worktree` thread is an independent +-owner with its own. Lane work needs the second. ++owner with its own. An independent task lane needs the second. + +-A **lane** is thread work; a **worker inside a lane** is subagent work. N lanes +-means N worktree threads, and the workers inside each lane are that lane's +-subagents — they cannot collide across lanes because the worktrees differ. ++An **independent task lane** owns a goal, PABCD cycle or long-running branch/CI ++lifecycle: use one worktree thread per lane. A **bounded checkout worker** needs ++only a disjoint checkout and returns a patch or evidence to the coordinator: ++create a managed worktree, then give its absolute path to a subagent. The ++subagent's native cwd still inherits the coordinator's; the packet must require ++that path as the shell workdir on every command. Workers do not acquire their ++own goal or PABCD state. Different workers must use different worktrees. +``` + +At table lines 35-37, make the cwd and edit visibility precise for the new subagent pattern: + +```diff +-| Working directory | the parent's, unchanged; never a copy | its own, with `environment: worktree`; the shared project checkout with `local` | ++| Working directory | native cwd inherits the parent's; a bounded worker must pass its assigned managed-worktree path as the shell workdir on every command | its own, with `environment: worktree`; the shared project checkout with `local` | +-| Git branch and HEAD | the parent's | its own under `worktree`; shared under `local` | ++| Git branch and HEAD | native cwd points at the parent's; commands run in an assigned managed worktree see that worktree's branch and HEAD | its own under `worktree`; shared under `local` | +-| Edits visible to the parent | immediately, as the parent's own uncommitted changes | only through git | ++| Edits visible to the parent | immediately in the selected checkout; a managed worktree has its own branch and files | only through git | +``` + +At lines 60-73, replace the first two rules and append the explicit workdir rule after the fork-context paragraph: + +```diff +-- Write scopes across concurrent subagents must not overlap. +-- **Never** run two subagents that perform branch-level git operations at the +- same time. `checkout`, `switch`, `branch`, `stash`, `reset`, `rebase`, `merge` +- and `pull` act on one shared HEAD; two children doing that corrupt each other's +- work regardless of how their file scopes were divided. A per-file write scope +- does not make concurrent branch work safe. ++- Write scopes across concurrent subagents must not overlap. A coordinator ++ assigning separate managed worktrees must give each worker a different path. ++- **Never** run concurrent branch-level git operations in one checkout. ++ `checkout`, `switch`, `branch`, `stash`, `reset`, `rebase`, `merge` and `pull` ++ change that checkout's HEAD or index; a per-file write scope does not separate ++ them. Operations in different worktrees do not share one HEAD, but each branch ++ still needs one owner and an explicit integration order. +@@ + Only the V2 usage hint states the shared directory, so a V1 session is never + told it by the runtime. ++- For a managed-worktree worker, instruct the subagent to pass the absolute ++ worktree path as the shell tool's workdir on **every** command, including ++ `git status`, tests and reads. Use absolute paths for file edits. Its native ++ cwd and relative-path defaults do not move when the worktree is created. +``` + +At lines 75-93, replace the route list with the following complete list. It states the `create_worktree` then `spawn_agent` sequence only for bounded checkout work; the independent-task route remains a thread: + +```diff + Route by what the work needs to own, not by how parallel it is: + +-- Needs its own branch, checkout, or long-running merge/CI lane -> **thread**, +- one per lane, created with `environment: worktree`. A `local` thread does not +- give the lane a checkout of its own. +-- Needs its own goal or its own PABCD cycle -> **thread**. +-- Is a bounded slice inside a lane that already owns its checkout -> **subagent** +- of that lane's thread. +-- Is a bounded slice of the tree you are already editing, returning evidence or a +- patch rather than owning a branch -> **subagent**. +-- Is read-only research -> **subagent**, by default. It cannot collide because it +- writes nothing, which is also why read-only fan-out is not a template for +- parallel write work. +- +-"Merge these lanes in parallel", "prepare N stacks at once", "run these branches +-concurrently" are thread work. Spawning N subagents for N branches puts N writers +-on one HEAD. ++- Needs its own goal, PABCD cycle, user-visible task, or long-running ++ merge/CI lifecycle -> **thread**, one per independent task lane, with ++ `environment: worktree` for an isolated checkout. A `local` thread shares ++ the checkout. ++- Needs an isolated checkout for a bounded, coordinator-owned write packet ++ while the coordinator is full-access -> call `create_worktree`, wait for its ++ completed absolute workspace path, then spawn a **subagent** with that path ++ and an instruction to pass it as the shell workdir on every command. Give ++ concurrent workers disjoint worktrees and prohibit concurrent branch ++ operations in one checkout. The coordinator owns goal/PABCD and integration. ++- Is a bounded slice inside a thread lane that already owns its checkout -> ++ **subagent** of that thread. ++- Is a bounded slice of the checkout you are already editing, returning ++ evidence or a patch -> **subagent** with disjoint file scope. ++- Is read-only research -> **subagent**, by default. Read-only fan-out is not ++ a template for parallel writes. ++ ++`create_thread` children may start with reduced approval permission, including ++projectless targets. Confirm their actual permission state before planning an ++unattended write lane. The bounded worktree/subagent route does not grant new ++permissions; it uses the coordinator's inherited subagent permission and an ++explicit checkout path. When a lane needs independent goal/PABCD ownership, ++keep the thread route and handle its actual permission state. +``` + +At lines 108-112, make the parallel-lane statement apply to independent tasks and add the bounded variant: + +```diff +-N independent lanes means N `worktree` threads, N checkouts, N FSMs. The parent +-coordinates with `wait_threads` and integrates; it does not advance any child's +-FSM, and a child does not advance the parent's. ++N independent task lanes mean N `worktree` threads, N checkouts and N FSMs. ++The coordinator uses `wait_threads` and integrates; neither side advances the ++other's FSM. N bounded checkout workers mean N managed worktrees and N ++subagents, with one coordinator goal/FSM. The coordinator uses the returned ++subagent handles and checks each worktree's files before integration. +``` + +At lines 201-206, keep the two composition forms distinct: + +```diff +-Threads and subagents then compose. A lane thread spawns its own subagents inside +-its own worktree, and subagents belonging to different lanes cannot collide +-**because those worktrees differ** — not because their parents are different +-tasks. Two `local` threads on one checkout collide exactly like two subagents do. +-The shape that scales is worktrees for isolation and subagents for concurrency +-within an isolated tree. ++Threads and subagents compose in independent task lanes: each worktree thread ++spawns bounded subagents inside its checkout. A full-access coordinator can ++also assign separate managed worktrees directly to bounded subagents. In both ++forms, different worktrees provide file and HEAD isolation; different thread ++ids do not. Two `local` threads on one checkout still collide. +``` + +At lines 208-213, change the manifest lead so it does not require a task id for a bounded worker: + +```diff +-Lanes are independent tasks, so nothing in the system knows two of them were handed the ++Independent task lanes are separate tasks, so nothing in the system knows two were handed the + same issue until their pull requests collide. One shared record makes that visible before + the branches diverge. Per lane: repository, lane id, task and host id, worktree, branch, + base ref and sha, head sha, issue, owner, scope and status. ++A bounded worktree worker remains under its coordinator and does not invent a ++thread id or its own FSM. Record its worktree path and assigned scope in the ++coordinator's packet or progress record instead. +``` + +The rest of the lane manifest and wake/poll guidance remains scoped to independent threads; the new paragraph above prevents applying its `threadId` requirement to bounded workers. + +### 2. MODIFY `plugins/codexclaw/skills/pabcd/references/delegation.md` + +At lines 3-6, qualify the opening description: + +```diff + packet. [Dispatch surfaces](dispatch-surfaces.md) owns the choice between a + subagent and a separate Codex task, and the fact that a subagent runs in this +-session's own working directory rather than a copy of it. ++session's native working directory rather than a copy of it. A bounded worker ++can operate in a separately created managed worktree only when its packet ++supplies that absolute path and it uses it as every shell command's workdir. +``` + +After the existing packet/scope instructions at lines 14-20, add this exact optional #265 section. `PROGRESS.md` is written only when the packet grants its path; a successor reads it plus the files and checks them before resuming. No new CLI or mandatory log is introduced. + +````md +### Optional worker progress checkpoint (#265) + +For a long bounded write packet, the coordinator may grant a specific `PROGRESS.md` +path inside the worker's assigned worktree. The worker may update it after a +coherent edit or check with three fields: `Done`, `Remaining`, and `Partial files` +(absolute paths plus what is incomplete). Example: + +```text +Done: parsed hook input and added the first regression test +Remaining: add manifest entries; run focused tests +Partial files: /absolute/worktree/path/src/agent-thread-permissions.ts — parser branch incomplete +``` + +The checkpoint is a handoff hint, not completion proof or a new source of +authority. On interruption, the coordinator checks that the first worker has +stopped, reads `PROGRESS.md` and the named files, then gives the replacement +worker the same bounded packet, worktree path, and remaining work. The +replacement verifies the actual file state before editing. Without a granted +path, the worker does not create `PROGRESS.md`. +```` + +At lines 207-225, add the permission observation after the thread-surface paragraph: + +```diff + `worktree` is what gives a lane its own checkout. Creating a thread is + user-visible; messaging one is not commanding it. ++A `create_thread` child may start with reduced approval permission even when ++the coordinator is full-access; this also occurs for projectless targets. Check ++the child's actual permission mode before assigning unattended writes. A ++bounded checkout worker can instead use `create_worktree` plus a subagent with ++the returned absolute path as every shell workdir. This does not give the ++subagent its own task, goal or PABCD state. +``` + +At lines 229-236, replace the isolation bullet: + +```diff +-- **DISPATCH-ISOLATION-01:** subagent lanes are not isolated environments — they +- all run in this session's working directory, so "isolation" here means scope +- discipline, not separation. Give every lane explicit read and write access lists +- with no overlap, and never share in-progress output across lanes. Concurrent +- lanes must never run branch-level git operations (`checkout`, `switch`, +- `branch`, `stash`, `reset`, `rebase`, `merge`, `pull`): those act on one shared +- HEAD and a per-file write scope does not make them safe. Work that genuinely +- needs its own branch or checkout is thread work, not a subagent lane. ++- **DISPATCH-ISOLATION-01:** subagents inherit the parent's native cwd; they ++ do not get a copied checkout. Give concurrent workers disjoint read/write ++ scopes. For a bounded worker in a managed worktree, assign one absolute ++ worktree path and require the shell workdir on every command; use absolute ++ file paths for edits. Different workers get different worktrees. Never run ++ concurrent branch-level operations (`checkout`, `switch`, `branch`, `stash`, ++ `reset`, `rebase`, `merge`, `pull`) in one checkout; a file scope cannot ++ separate one HEAD. Work that needs its own goal or PABCD cycle stays a ++ separate thread task. +``` + +## Activation scenarios and verification + +1. Full-access coordinator, two bounded write packets: create two managed worktrees, wait for both absolute workspace paths, spawn two subagents with disjoint scopes, and require each shell command's `workdir` to equal its assigned path. Confirm `pwd`, `git rev-parse --show-toplevel`, branch/HEAD and edits in each worktree. Their native inherited cwd is not evidence of the command workdir. No simultaneous branch operation in one checkout. +2. Independent issue lane needing its own goal/PABCD: use a worktree thread, confirm its actual permission mode, and retain `wait_threads`, lane manifest, and separate FSM ownership. The managed-worktree/subagent route cannot satisfy this case. +3. Interrupted bounded worker with a granted checkpoint path: ensure it stopped, inspect `PROGRESS.md` and partial files, then dispatch a replacement with the same exact path/scope. Without a granted path, no checkpoint is created. These are manual acceptance scenarios for guidance, not claims of automated enforcement. + +Run `rg -n 'native cwd|shell workdir|create_worktree|PROGRESS.md|concurrent branch' plugins/codexclaw/skills/pabcd/references/{dispatch-surfaces,delegation}.md` and inspect the modified paragraphs for contradictory unconditional statements. Run `git diff --check -- plugins/codexclaw/skills/pabcd/references/dispatch-surfaces.md plugins/codexclaw/skills/pabcd/references/delegation.md`. No executable code changed in this unit; do not claim the permission behavior itself has been tested by these text checks. + +## Out of scope + +No new CLI, automatic worker checkpoint, changed thread-creation API, permission grant, branch automation, or goal/FSM handoff to a subagent. The coordinator remains responsible for verifying worker files and integrating results. diff --git a/devlog/_plan/260927_issue_train/030_wp4_goalplan_decisions.md b/devlog/_plan/260927_issue_train/030_wp4_goalplan_decisions.md new file mode 100644 index 00000000..48b16bea --- /dev/null +++ b/devlog/_plan/260927_issue_train/030_wp4_goalplan_decisions.md @@ -0,0 +1,300 @@ +# wp4 — Goalplan decisions for issue #262 + +Record a submitted user question in the goalplan and link only the affected work phases to it. An open linked decision makes those phases unavailable to `ready`, cursor selection, task completion, and close/recovery; recording an answer restores their ordinary readiness. The `ask` verb records a question **after** the host sends it. It never sends a message or supplies an answer. This unit uses the delegated contract (`open|decided`, free-text answer and optional recommendation); issue #262's earlier `options[]`/`withdrawn` sketch is outside this unit. + +## Phase contract + +- Work phase: `wp4`, issue [#262](https://github.com/lidge-jun/codexclaw/issues/262). Docs-only implementation record; builder changes code in a later phase. +- Class: C3, with C4 care for the scheduler/enforcement paths. No schema-version bump: both fields are opt-in and absent on old plans. +- Completion: an open decision hides only its linked phases, including their tasks and explicit cursor; `decide` releases the wait without changing phase status or clearing `blockedReason`; malformed decision data fails closed; old plans round-trip without acquiring new fields. +- E8 decision: `validateGoalplan` fails while a linked phase is pending/in progress because that phase remains unfinished. An unrelated open question does not veto an otherwise complete plan. A phase marked `done` while it still awaits an open decision is an integrity error, so hand editing cannot certify completion by bypassing the wait. This follows the rule that only the dependent action stays pending (`plugins/codexclaw/skills/dev/references/async-questions.md:43-60`) and the existing remaining-work gate (`plugins/codexclaw/components/pabcd-state/src/goalplan.ts:1491-1494`). + +## File change map + +All paths below are repository-relative. Anchors were rechecked against `codex/issue-train-0927` at `958441a9` on 2026-09-27. No new source module, separate decision store, or Interview ledger change. + +### 1. MODIFY `plugins/codexclaw/components/pabcd-state/src/goalplan.ts` + +The phase schema is at `:124-136`; the plan schema at `:216-236`; the explicit reviver is at `:507-606`; invalid-shape diagnostics are at `:806-843`; atomic publication and lock are at `:737-861`. Add the following types/fields, with no defaults on `buildGoalplan`: + +```diff + export interface GoalplanWorkPhase { + // existing fields + dependsOn?: string[]; ++ /** Decision ids; any open target pauses this phase without changing status. */ ++ awaitsDecision?: string[]; + blockedReason?: string; + } ++export interface GoalplanDecision { ++ id: string; ++ question: string; ++ recommendation?: string; ++ status: "open" | "decided"; ++ answer?: string; ++ askedAt: string; ++ decidedAt?: string; ++} + export interface Goalplan { + // existing fields + workPhases: GoalplanWorkPhase[]; ++ decisions?: GoalplanDecision[]; + } +``` + +Use the existing `LIFECYCLE_ID_RE` at `:1120` for decision ids. Add this complete structural reviver before `reviveGoalplan` at `:507`. It rejects malformed optional fields rather than dropping them; otherwise a stored wait could vanish when the plan is read. Absence and `[]` stay distinct. Keep the exact string values on disk; use trimming only to test emptiness. + +```ts +function validIsoTime(value: unknown): value is string { + if (typeof value !== "string") return false; + const date = new Date(value); + return Number.isFinite(date.valueOf()) && date.toISOString() === value; +} + +function reviveDecisions(value: unknown): GoalplanDecision[] | undefined | "invalid" { + if (value === undefined) return undefined; + if (!Array.isArray(value)) return "invalid"; + const decisions: GoalplanDecision[] = []; + for (const item of value) { + if (typeof item !== "object" || item === null || Array.isArray(item)) return "invalid"; + const d = item as Record; + if (typeof d.id !== "string" || !LIFECYCLE_ID_RE.test(d.id) + || typeof d.question !== "string" || !d.question.trim() + || !validIsoTime(d.askedAt) + || (d.recommendation !== undefined && (typeof d.recommendation !== "string" || !d.recommendation.trim()))) return "invalid"; + if (d.status === "open") { + if (d.answer !== undefined || d.decidedAt !== undefined) return "invalid"; + decisions.push({ id: d.id, question: d.question, status: "open", askedAt: d.askedAt, + ...(d.recommendation === undefined ? {} : { recommendation: d.recommendation as string }) }); + } else if (d.status === "decided") { + if (typeof d.answer !== "string" || !d.answer.trim() || !validIsoTime(d.decidedAt)) return "invalid"; + decisions.push({ id: d.id, question: d.question, status: "decided", answer: d.answer, + askedAt: d.askedAt, decidedAt: d.decidedAt, + ...(d.recommendation === undefined ? {} : { recommendation: d.recommendation as string }) }); + } else return "invalid"; + } + return decisions; +} +``` + +At `:525-550`, parse `w.awaitsDecision` with the same strict nonempty-string-array rule as `reviveDependsOn`; reject `"invalid"`, then attach the value to `phase` only when present. At `:581-605`, call `reviveDecisions(o.decisions)`, reject `"invalid"`, and attach only when present. At `firstInvalidField` (`:820-842`), report `"workPhases[].awaitsDecision"` and `"decisions"` in the same order as the reviver. For a malformed decision entry, report `"decisions"`, not `(unknown)`. + +Add these pure helpers beside `isRunnablePhase` at `:983` and replace its body. Missing references fail closed in selection even if the caller skipped integrity validation: + +```ts +export function openDecisionIdsForPhase(plan: Goalplan, wp: GoalplanWorkPhase): string[] { + return [...new Set(wp.awaitsDecision ?? [])].filter((id) => + plan.decisions?.find((decision) => decision.id === id)?.status !== "decided"); +} + +function workPhaseReadyConditionsMet(plan: Goalplan, wp: GoalplanWorkPhase): boolean { + return workPhaseDependenciesMet(plan, wp) && openDecisionIdsForPhase(plan, wp).length === 0; +} + +function isRunnablePhase(plan: Goalplan, wp: GoalplanWorkPhase): boolean { + return (wp.status === "pending" || wp.status === "in_progress") + && workPhaseReadyConditionsMet(plan, wp); +} +``` + +`readyWorkPhases` and `readyTasks` then inherit the gate at `:990-1004`. In `dependencyWaitReasons` (`:1050-1072`), after the existing phase-dependency reason and before the task loop, append `work-phase ${wp.id} awaits decision ${ids.join(", ")}` for nonempty `openDecisionIdsForPhase`, including when independent work is ready. In `dependencyDeadlock` (`:1078-1117`), append the same decision reason for a waiting phase before entering its task loop; keep the existing explicit-blocked reason first for `status === "blocked"`. A phase waiting on both a prerequisite and a decision should report both direct reasons; do not let the prerequisite branch `continue` before the decision reason. Preserve the current dependency/task reason strings for legacy plans. + +At `goalplanDefinitionIntegrityReasons` (`:1311-1387`), add exact checks: duplicate decision ids; duplicate ids within one `awaitsDecision`; unknown decision ids; and a `done` phase that awaits an open decision. Example messages: `duplicate decision id 'dec-1' makes awaitsDecision references ambiguous`, `work phase wp-a awaits unknown decision 'dec-missing'`, and `work phase wp-a is done while decision dec-1 is open`. Put these before task/criterion diagnostics so `ready` and E8 expose the broken reference promptly. `ready` already calls this function before returning data (`goalplan-cli.ts:427-437`). + +```ts +const decisionsById = new Map((plan.decisions ?? []).map((decision) => [decision.id, decision])); +for (const id of duplicateIds((plan.decisions ?? []).map((decision) => decision.id))) { + reasons.push(`duplicate decision id '${id}' makes awaitsDecision references ambiguous`); +} +for (const phase of plan.workPhases) { + for (const id of duplicateIds(phase.awaitsDecision ?? [])) { + reasons.push(`work phase ${phase.id} awaits decision '${id}' more than once`); + } + for (const id of new Set(phase.awaitsDecision ?? [])) { + const decision = decisionsById.get(id); + if (!decision) reasons.push(`work phase ${phase.id} awaits unknown decision '${id}'`); + else if (phase.status === "done" && decision.status === "open") { + reasons.push(`work phase ${phase.id} is done while decision ${id} is open`); + } + } +} +``` + +Add the following transitions beside `addGoalplanTask` (`:1127-1171`). They are pure; the CLI owns the locked read/write. The `ask` duplicate check applies only to still-open questions and compares trimmed question text exactly. Reusing a decided question's text is allowed as a new id. Do not mutate phase status or `blockedReason` on `decide`. + +```ts +export function askGoalplanDecision( + plan: Goalplan, + input: { id: string; question: string; recommendation?: string; workPhaseIds: string[]; askedAt: string }, +): GoalplanLifecycleResult { + const id = input.id.trim(); + const question = input.question.trim(); + const recommendation = input.recommendation?.trim(); + const workPhaseIds = input.workPhaseIds.map((phaseId) => phaseId.trim()); + if (!LIFECYCLE_ID_RE.test(id)) return { kind: "rejected", reason: "decision id must be a short lowercase id, e.g. dec-1" }; + if (!question) return { kind: "rejected", reason: "decision question must not be empty" }; + if (input.recommendation !== undefined && !recommendation) return { kind: "rejected", reason: "decision recommendation must not be empty" }; + if (!validIsoTime(input.askedAt)) return { kind: "rejected", reason: "decision askedAt must be an ISO timestamp" }; + if (plan.decisions?.some((decision) => decision.id === id)) return { kind: "rejected", reason: `decision '${id}' is already in this plan` }; + const duplicate = plan.decisions?.find((decision) => decision.status === "open" && decision.question.trim() === question); + if (duplicate) return { kind: "rejected", reason: `question is already open as decision '${duplicate.id}'` }; + if (workPhaseIds.some((phaseId) => !phaseId) || new Set(workPhaseIds).size !== workPhaseIds.length) { + return { kind: "rejected", reason: "--work-phase requires distinct non-empty ids" }; + } + for (const phaseId of workPhaseIds) { + const phase = plan.workPhases.find((wp) => wp.id === phaseId); + if (!phase) return { kind: "rejected", reason: `work phase '${phaseId}' is not in this plan` }; + if (phase.status === "done" || phase.status === "superseded") { + return { kind: "rejected", reason: `work phase '${phaseId}' is ${phase.status} and cannot await a decision` }; + } + } + const decision: GoalplanDecision = { id, question, status: "open", askedAt: input.askedAt, + ...(recommendation === undefined ? {} : { recommendation }) }; + const next: Goalplan = { ...plan, decisions: [...(plan.decisions ?? []), decision], + workPhases: plan.workPhases.map((wp) => workPhaseIds.includes(wp.id) + ? { ...wp, awaitsDecision: [...(wp.awaitsDecision ?? []), id] } : wp) }; + const reasons = goalplanDefinitionIntegrityReasons(next); + return reasons.length ? { kind: "rejected", reason: reasons.join("; ") } : { kind: "changed", plan: next }; +} + +export function decideGoalplanDecision( + plan: Goalplan, id: string, answer: string, decidedAt: string, +): GoalplanLifecycleResult { + const decision = plan.decisions?.find((candidate) => candidate.id === id.trim()); + if (!decision) return { kind: "rejected", reason: `decision '${id.trim()}' is not in this plan` }; + if (!answer.trim()) return { kind: "rejected", reason: "decision answer must not be empty" }; + if (!validIsoTime(decidedAt)) return { kind: "rejected", reason: "decision decidedAt must be an ISO timestamp" }; + if (decision.status === "decided") return decision.answer === answer.trim() + ? { kind: "unchanged", plan, reason: `decision '${id.trim()}' is already decided` } + : { kind: "rejected", reason: `decision '${id.trim()}' already has a different answer` }; + const next: Goalplan = { ...plan, decisions: plan.decisions!.map((candidate) => candidate.id === decision.id + ? { ...candidate, status: "decided" as const, answer: answer.trim(), decidedAt } : candidate) }; + return { kind: "changed", plan: next }; +} +``` + +Enforcement bypass record: `effectiveActiveWorkPhaseId` falls back to raw dependency checks at `:2043-2049`; replace both fallback predicates with `isRunnablePhase(plan, wp)`. `closeFixedWorkPhase` checks only prerequisites at `:1795` and its successor search/recorded-marker branches at `:1822-1877`; make the following exact substitutions: + +```diff +-if (!workPhaseDependenciesMet(plan, current)) { +- return { kind: "dependencies_unmet", unmet: unmetPhaseDependencyIds(plan, current) }; ++if (!workPhaseReadyConditionsMet(plan, current)) { ++ return { kind: "dependencies_unmet", unmet: [ ++ ...unmetPhaseDependencyIds(plan, current), ++ ...openDecisionIdsForPhase(plan, current).map((id) => `decision:${id}`), ++ ] }; + } + // In BOTH after/wrap successor finds at :1822-1827: +-wp.status === "pending" && workPhaseDependenciesMet(closedPlan, wp) ++isRunnablePhase(closedPlan, wp) + // For the retained in-progress cursor at :1869-1872: +-wp.status === "in_progress" && workPhaseDependenciesMet(closedPlan, wp) ++wp.status === "in_progress" && workPhaseReadyConditionsMet(closedPlan, wp) + // For a named successor at :1875: +-!workPhaseDependenciesMet(closedPlan, named) ++!workPhaseReadyConditionsMet(closedPlan, named) + // For resumeAbsentTarget at :1954: +-!workPhaseDependenciesMet(plan, named) ++!workPhaseReadyConditionsMet(plan, named) + // For effectiveActiveWorkPhaseId at :2043-2049, in both finds: +-wp.status === "in_progress" && workPhaseDependenciesMet(plan, wp) ++wp.status === "in_progress" && isRunnablePhase(plan, wp) +-wp.status === "pending" && workPhaseDependenciesMet(plan, wp) ++wp.status === "pending" && isRunnablePhase(plan, wp) +``` + +Keep the existing `dependencies_unmet` and `successor_lost` result variants; callers already handle them. Change `absentSuccessorDetail` at `:1918-1925` to say `now waits for a prerequisite or decision`. A recovered marker may never activate a newly waiting phase. This is necessary because `advanceWorkPhase` calls `closeFixedWorkPhase` at `:2012`, and hooks/orchestrator derive their target from `effectiveActiveWorkPhaseId` (`hook.ts:346`, `orchestrate-cli.ts:83`). No hook-specific filter is needed once these common helpers are fixed. + +### 2. MODIFY `plugins/codexclaw/components/pabcd-state/src/goalplan-cli.ts` + +Imports are at `:14-38`; parser and per-verb flag rules are at `:51-235`; locked lifecycle mutation is at `:483-588`; `ready` at `:427-470`; `show` at `:591-614`; dispatch/help at `:621-739`. Import `askGoalplanDecision`, `decideGoalplanDecision`, `openDecisionIdsForPhase`, and `goalplanDefinitionIntegrityReasons` (already imported). Extend the existing parser, without allowing these flags on other verbs: + +```diff + export type GoalplanVerb = + // existing verbs ++ | "ask" | "decide" + export interface GoalplanCliArgs { + // existing fields ++ question?: string; ++ recommendation?: string; ++ answer?: string; ++ workPhaseIds?: string[]; + } + type GoalplanFlag = + // existing flags ++ | "--question" | "--recommendation" | "--answer"; + const VERBS = new Set([ + // existing verbs ++ "ask", "decide", + ]); +``` + +Add verb rules: `ask` allows `--session`, `--id`, `--question`, `--recommendation`, repeated `--work-phase`, `--cwd`; `decide` allows `--session`, `--id`, `--answer`, `--cwd`. Their usage strings are exactly `ask --session --id --question [--recommendation ] [--work-phase ]... [--cwd ]` and `decide --session --id --answer [--cwd ]`. Update the unknown-verb diagnostic at `:164-167`, help's verb list at `:626`, and help notes at `:629-641`. In the parser initializer at `:171`, add `workPhaseIds: []`; add cases for the three new value flags and, for `--work-phase`, push trimmed distinct nonempty ids only when `selected === "ask"`, otherwise retain its current singleton `workPhaseId` behavior. The verb rule handles wrong-verb flags; the existing missing-value and repeat rules at `:182-204` handle syntax before any write. + +Add `runDecision` before `runLifecycle`; dispatch it before slug-based reads at `:712-720`. Follow the same canonical-session and bound-slug checks as `runLifecycle` (`:483-498`), including the same no-write error. Validate all required fields before entering the lock. Then use this exact locked transition shape: + +```ts +type DecisionCommit = { kind: "rejected"; reason: string } | { kind: "changed" } | { kind: "unchanged"; reason: string }; +const locked = withGoalplanWriteLock(args.cwd, slug, (plan) => { + const result = args.verb === "ask" + ? askGoalplanDecision(plan, { + id, question: args.question!, recommendation: args.recommendation, + workPhaseIds: args.workPhaseIds ?? [], askedAt: new Date().toISOString(), + }) + : decideGoalplanDecision(plan, id, args.answer!, new Date().toISOString()); + if (result.kind === "rejected") return { kind: "rejected", reason: result.reason }; + if (result.kind === "unchanged") return { kind: "unchanged", reason: result.reason }; + writeGoalplan(args.cwd, result.plan); + return { kind: "changed" }; +}); +``` + +Map `locked`/`unreadable` to code 1 and `loop ${args.verb}: ${locked.reason}`; rejected to code 1 with its reason; unchanged to code 0 with `nothing to do`; changed to code 0 naming id and slug. Do not append an Interview event or claim the question was sent. `withGoalplanWriteLock` already reads under lock and returns `unreadable` for a bad plan (`goalplan.ts:737-804`); `writeGoalplan` performs the atomic write (`goalplan.ts:845-861`). `ask` is intentionally separate from `applySteeringBatch`, which only accepts additive criterion/work-phase operations (`goalplan-cli.ts:343-405`). + +For `ready`, retain `readyWorkPhases`/`readyTasks` output and add `openDecisions` (id, question, recommendation, askedAt) and `awaitingDecisions` (workPhaseId, decisionIds) in JSON, plus readable lines in text. Include the new keys/lines only when `plan.decisions !== undefined`, keeping the old plan's JSON/text shape unchanged. A linked phase appears in `awaitingDecisions` only while its status is pending/in_progress and at least one referenced decision is open. `show` lists each open decision and a `waiting: wp-a, wp-b` line derived from `awaitsDecision`; keep explicit blocked phases in their existing status display. Neither command mutates the plan. + +### 3. MODIFY `plugins/codexclaw/skills/loop/references/durable-goalplan.md` + +At schema bullets `:45-77`, add optional `decisions[]` and `workPhase.awaitsDecision?: string[]` with the exact shape above and explicit absent/empty compatibility. At CLI bullets `:79-104`, document `ask` and `decide`, the required ordering (send via the host's question tool, then record with `ask`; on a user reply record with `decide`), and that `ask` never sends. Document `ready`'s `openDecisions`/`awaitingDecisions` and `show`'s waiting display; explain that `decide` only changes the decision record. Do not state that every open decision blocks the whole goal. + +### 4. MODIFY `plugins/codexclaw/skills/dev/references/async-questions.md` + +After `:43-49`, add: “For a bound goalplan, after the host confirms a question was submitted, run `cxc loop ask --session --id --question [--recommendation ] [--work-phase ]...` to keep it in the existing plan. Record each dependent work phase. Check `cxc loop show` before sending a similar question after compaction. When the user answers, run `cxc loop decide --session --id --answer ` and resume only newly runnable phases. `ask` does not deliver a question; a CLI success is no proof of host submission.” Keep the optional-answer assumption rule at `:50-55`: link a phase only while its action truly requires the reply. Keep Interview separate (`:69`). + +### 5. MODIFY `plugins/codexclaw/components/pabcd-state/test/goalplan-public-surface.test.ts` + +Reuse `fixture`, `workspace`, `planText`, `ledgerText`, and `cli` at `:18-72`. Add these named tests with exact assertions (each goes red before this unit because the verb/field or readiness rule is absent): + +1. `ask records an open decision and hides only linked work phases` — add independent `wp-free` to the fixture; call `ask --session sess-public --id dec-1 --question "Choose API" --recommendation "Use v2" --work-phase wp-live`. Assert exit 0, stored decision has open status and ISO `askedAt`, `wp-live.awaitsDecision === ["dec-1"]`, `wp-free` remains in `readyWorkPhases`, `wp-live` and its tasks do not; `ready --json` contains `openDecisions[0].id === "dec-1"` and `awaitingDecisions` names `wp-live`; `show` names the question and waiting phase. This tests the live cursor too: `effectiveActiveWorkPhaseId` must choose `wp-free`. +2. `decide releases linked phases without unblocking explicit blocks` — start with one pending linked phase and one `status: "blocked"` linked phase with `blockedReason: "vendor"`; ask and then decide with a nonempty answer. Assert `answer`, `decidedAt`, and `status === "decided"`; the pending phase is ready, the blocked phase remains blocked with its original reason, and no `awaitingDecisions` entry remains. +3. `ask rejects duplicate open question and unknown phase without a write` — after a valid ask, snapshot `planText` and `ledgerText`; retry same trimmed question under `dec-2`, then try a distinct question with `--work-phase ghost`. Assert both code 1, diagnostic names `dec-1`/`ghost` respectively, and both files are byte-identical to snapshots. +4. `ask and decide enforce per-verb flags before writing` — parse wrong-verb `--answer`/`--question`, duplicate singleton flags, duplicate `--work-phase`, empty `--question=`, and missing value; assert `error` and unchanged plan/ledger. Check help contains both exact usage strings, and update the unknown-verb expected list at `:494-496`. +5. `decide is idempotent only for the same answer` — second identical answer returns code 0 and byte-identical plan; different answer returns code 1 and byte-identical plan. This guards against silent answer replacement after compaction. + +### 6. MODIFY `plugins/codexclaw/components/pabcd-state/test/work-phase-states.test.ts` + +Reuse `phase`, `plan`, and `roundTrip` at `:31-40` and `:155-161`. Add these named tests: + +1. `open decision excludes linked phase from cursor close and successor selection` — use a linked `in_progress` cursor and an independent pending phase; assert `effectiveActiveWorkPhaseId` selects the independent phase, `advanceWorkPhase` cannot close the linked one, and after the independent phase closes the linked one is still pending/in_progress and `dependencyDeadlock()?.reasons` names the decision. Also cover `closeFixedWorkPhase` and `resumeAbsentTarget` with a recorded linked successor; both must refuse activation. +2. `decision wait reasons appear beside ready independent work` — with a free pending phase and a linked pending phase, assert `dependencyDeadlock() === null` and `dependencyWaitReasons()` contains `work-phase linked awaits decision dec-1`. +3. `invalid decision fields and dangling references fail closed` — import `readGoalplanDetailed` and `goalplanDefinitionIntegrityReasons`. Hand edit stored JSON to put a malformed `decisions`, malformed `awaitsDecision`, then a valid array referencing `ghost`; assert malformed variants give `readGoalplan(...) === null` with `readGoalplanDetailed(...).diagnostic.field` naming the field, while a dangling reference yields a definition-integrity reason and no linked phase in `readyWorkPhases`. Add a `ready` CLI assertion to `goalplan-public-surface.test.ts` using its existing `cli` helper: code 1 and the same unknown-decision diagnostic. Repeat for duplicate decision id and duplicate phase reference. +4. `legacy plan round trips without decision fields` — compare the old plan and read-back plan excluding `updatedAt` (which `writeGoalplan` always refreshes at `goalplan.ts:858`); assert `"decisions" in back === false`, `"awaitsDecision" in back.workPhases[0] === false`, and the old ready/validation result is unchanged. This extends the existing legacy case at `:204-209`. +5. `E8 fails a done phase waiting on an open decision but permits an unrelated open decision` — with otherwise complete schema-v1 fixtures, assert the linked/done plan has an integrity reason; assert an unlinked open decision leaves `validateGoalplan(...).ok === true`. A linked pending phase must fail through the existing `work phase(s) not done` reason. + +These tests must include the negative enforcement paths because a filtered `ready` result alone cannot prove that cursor or recovery cannot advance a linked phase (`goalplan.ts:1795-1877`, `:1933-1975`, `:2033-2050`). Run focused tests with `node plugins/codexclaw/scripts/test.mjs "plugins/codexclaw/components/pabcd-state/test/goalplan-public-surface.test.ts" "plugins/codexclaw/components/pabcd-state/test/work-phase-states.test.ts"`, then `npm run build` (the root `package.json:21-24` defines build/test but no standalone typecheck). Run `git diff --check` after implementation. These are future builder checks; this docs-only phase did not run code tests. + +## Activation scenarios + +| Condition | Trigger and expected path | +| --- | --- | +| No decision fields | Old plan loads without synthetic fields; ready, show, E8, and write/read keep prior behavior. | +| Open, unlinked decision | Listed in `openDecisions`; independent work and E8 completion continue. | +| Open, linked decision | `ready`, tasks, cursor, successor, close and recovery refuse the linked phase; other phases continue; waits are visible even when other work is ready. | +| Two waits on one phase | Either open id keeps it waiting; deciding one leaves the other wait; deciding both restores ordinary dependency/status checks. | +| Decision answered | Store answer/time; linked phase becomes eligible only if its normal prerequisites and status permit; explicit `blocked` remains blocked. | +| Bad stored reference/shape | Structural corruption fails read; unknown/duplicate references fail integrity and `ready`/E8, and selection fails closed. | +| Repeated question/answer | Open duplicate question or conflicting decided answer is rejected without a write; same decided answer is a no-op. | +| Host submission absent | Do not run `ask`; CLI does not contact the host or imply an ask occurred. | + +## Out of scope + +No host question tool integration, automatic answer capture, reminder/polling mechanism, `options[]` validation, `withdrawn` status, global goal pause, schema migration/version bump, Interview ledger reuse, or new standalone decision file. Issue #262's historical option/recommendation proposal differs from this phase's delegated free-text contract; if the parent wants the issue's options semantics, it needs a separate contract decision before implementation. diff --git a/devlog/_plan/260927_issue_train/040_wp5_delivery.md b/devlog/_plan/260927_issue_train/040_wp5_delivery.md new file mode 100644 index 00000000..73d0d0a4 --- /dev/null +++ b/devlog/_plan/260927_issue_train/040_wp5_delivery.md @@ -0,0 +1,24 @@ +# wp5 — Delivery and issue disposition + +This phase lands nothing new; it proves the merged state and records the issue decisions. + +## Per implementation phase (wp2, wp3, wp4) + +1. Branch `codex/issue-train-wpN` from current `origin/dev` (wp2 starts from this session's `codex/issue-train-0927`, which equals `origin/dev` 958441a9 plus this unit's docs). +2. Local gates at the phase C: `npm run build`, focused tests through `cxc receipt test`, `npm test` (record the TAP total), `node plugins/codexclaw/scripts/inventory.mjs --check --tests `, `node plugins/codexclaw/scripts/gate.mjs`, `node plugins/codexclaw/scripts/platform-smoke.mjs`. +3. Privacy self-check before first push (DEV-PRIVACY-01): grep the push range for client names, personal paths beyond this user's own home, tokens. +4. `git push -u origin codex/issue-train-wpN`; `gh pr create --base dev` with a body file (problem, behavior before/after, tests, residual risk, `Fixes #n` only for issues fixed in that PR). +5. Hosted CI: confirm ci.yml jobs actually ran against the PR head SHA (`gh pr view --json headRefOid,statusCheckRollup`, `gh run list --commit `), distinguishing pending, skipped, cancelled and failed. +6. Merge with the repository's usual method for dev PRs (squash, matching #245-#249), then rebase the next phase branch onto the new `origin/dev`. + +## Issue disposition + +- Fixed issues close automatically through `Fixes #n`; verify each closed. +- Deferred issues get one comment linking `devlog/_plan/260927_issue_train/001_research.md` on dev and the reason line from the table; they stay open. +- #261 gets the decline reason and is closed as not planned. +- #265 gets a comment linking the guidance PR; it stays open for the deferred CLI. + +## Acceptance + +All goalplan criteria met with captured evidence; `cxc loop validate` passes; `origin/dev` contains the three merge commits; no PR into `main` was opened by this unit. + From 09e9cb799e7e6c01bf762fee6a8a6b14f58b1a71 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 00:05:30 +0900 Subject: [PATCH 03/90] docs(plan): fold audit rounds 1-3 and re-plan #255 as gitignore partial fix --- devlog/_plan/260927_issue_train/000_plan.md | 17 ++- .../_plan/260927_issue_train/001_research.md | 7 +- .../002_architect_consultation.md | 10 +- .../010_wp2_runtime_overview.md | 16 +-- .../011_issue255_codexclaw_gitignore.md | 51 +++++++++ .../011_issue255_lazy_session_state.md | 106 ------------------ .../012_issue252_pabcd_switch.md | 24 +++- .../013_issue253_idle_goal_release.md | 4 +- .../014_issue254_turn_budget.md | 8 +- .../015_issue251_worker_gate.md | 14 ++- .../016_issue250_trigger_narrowing.md | 30 +++-- .../020_wp3_agent_thread_permissions.md | 6 +- .../030_wp4_goalplan_decisions.md | 4 + 13 files changed, 142 insertions(+), 155 deletions(-) create mode 100644 devlog/_plan/260927_issue_train/011_issue255_codexclaw_gitignore.md delete mode 100644 devlog/_plan/260927_issue_train/011_issue255_lazy_session_state.md diff --git a/devlog/_plan/260927_issue_train/000_plan.md b/devlog/_plan/260927_issue_train/000_plan.md index 5e2c347e..84d60a0b 100644 --- a/devlog/_plan/260927_issue_train/000_plan.md +++ b/devlog/_plan/260927_issue_train/000_plan.md @@ -1,6 +1,6 @@ # Issue train 2026-09-27: hook runtime fixes, agent-thread permissions, goalplan decisions -Codexclaw has four confirmed defects in its hook runtime, one host workaround worth shipping, and three small opt-in improvements among the 22 open issues. The defects are: ordinary prompt words inject PABCD directives (#250), the SubagentStop evidence gate blocks Codex's built-in `worker` in sessions that never used PABCD (#251), the Stop hook blocks every session with an active native goal even when no goalplan is bound (#253), and SessionStart writes `.codexclaw/sessions/.json` into every working directory (#255). The host workaround covers threads that Codex Desktop creates through `create_thread` with reduced permission even when the user runs full access. This unit fixes the defects, adds the opt-in permission hook and advisory, adds a PABCD off switch (#252), a per-turn Stop budget (#254) and plan-local pending decisions (#262), and records a triage decision for every open issue (001). +Codexclaw has four confirmed defects in its hook runtime, one host workaround worth shipping, and three small opt-in improvements among the 22 open issues. The defects are: ordinary prompt words inject PABCD directives (#250), the SubagentStop evidence gate blocks Codex's built-in `worker` in sessions that never used PABCD (#251), the Stop hook blocks every session with an active native goal even when no goalplan is bound (#253), and SessionStart creates unignored `.codexclaw` state in fresh working directories (#255, partial fix). The host workaround covers threads that Codex Desktop creates through `create_thread` with reduced permission even when the user runs full access. This unit fixes #250/#251/#253, partially fixes #255 by adding a `.gitignore` at first directory creation, adds the opt-in permission hook and advisory, adds a PABCD off switch (#252), a per-turn Stop budget (#254) and plan-local pending decisions (#262), and records a triage decision for every open issue (001). Reader: a maintainer deciding whether to merge these changes into dev; familiarity with the pabcd-state hook component and the goalplan CLI is assumed. @@ -8,7 +8,7 @@ Reader: a maintainer deciding whether to merge these changes into dev; familiari - Loop archetype: satisfy-spec HOTL, docs-first (LOOP-DOCS-FIRST-01). - Trigger: user request on 2026-09-27 to fix the real issues and worthwhile improvements among the open issues, plus the agent-thread permission problem found in the same chat, and put them into dev, through cxc-loop with unlimited gpt-6-sol dispatch. -- Goal: wp2-wp4 merged into `dev` through ordinary PRs after hosted CI; every open issue has a recorded decision; fixed issues are closed with PR links. +- Goal: wp2-wp4 merged into `dev` through ordinary PRs after hosted CI; every open issue has a recorded decision; fully fixed issues are closed with PR links; #255 stays open with the partial-fix PR linked and lazy creation deferred. - Non-goals: dev to main promotion, release, version bump, tags, npm publish; Codex core or Desktop changes; other repositories; the deferred and declined proposals in 001. - Verifier: per-phase focused `node --test` files named in each decade doc; at every C, `npm run build`, the focused tests through `cxc receipt test`, then full `npm test`, `node plugins/codexclaw/scripts/gate.mjs`, `node plugins/codexclaw/scripts/inventory.mjs --check --tests ` and `node plugins/codexclaw/scripts/platform-smoke.mjs`; hosted CI on each PR head (jobs actually ran, head SHA, event, run id). Skill prose changes are read by no test; their review is human (PLAN-VERIFIER-REAL-01). - Stop condition: all seven goalplan criteria met with fresh evidence, or a real blocker after root-cause work. @@ -21,9 +21,9 @@ Reader: a maintainer deciding whether to merge these changes into dev; familiari IN (details in each decade doc): ``` -plugins/codexclaw/components/pabcd-state/{src,dist,test} wp2 (010-016), wp3 (020), wp4 (030) -plugins/codexclaw/components/cxc-ops/{src,dist,test} wp2 (011 SessionStart writes), wp3 (hook-trust) -plugins/codexclaw/components/recall, bg-wake (SessionStart only) wp2 (011) where a SessionStart write is found +plugins/codexclaw/components/pabcd-state/{src,dist,test} wp2 (010 overview, 011_issue255_codexclaw_gitignore.md, 012-016), wp3 (020), wp4 (030) +plugins/codexclaw/components/cxc-ops/{src,dist,test} wp2 (011 first-directory writers), wp3 (hook-trust) +plugins/codexclaw/components/bg-wake, subagent-config, messenger-bridge/{src,dist,test} wp2 (011 cwd first-directory writers) plugins/codexclaw/hooks/*.json, .codex-plugin/plugin.json wp3 (two new hooks) plugins/codexclaw/skills/{pabcd,loop,dev}/references/*.md wp2 (015, 012), wp3 (021), wp4 (030) plugins/codexclaw/inventory.json, README family counts every phase that changes tests or hooks @@ -34,8 +34,8 @@ OUT: Codex core/Desktop, host automation mutation handler (#213), SessionStart f ## Ordered work phases - wp1 (this cycle): docs-only roadmap: 001 triage, 002 architect consultation, 010-016, 020-021, 030, 040. No production edits. -- wp2: hook runtime fixes (010 overview; 011 #255 lazy session state, 012 #252 PABCD switch, 013 #253 IDLE goal release, 014 #254 per-turn Stop budget, 015 #251 worker gate, 016 #250 trigger narrowing). Foundation first: 011 changes how state comes to exist and 012 adds the switch every later PABCD handler consults; 013 and 014 then change Stop continuation; 015 and 016 are leaf policy changes. Built in parallel by gpt-6-sol builders on separate branches in task-owned worktrees, merged in that order. -- wp3: agent-created thread permissions (020 hook and advisory, 021 dispatch guidance plus #265 checkpoint guidance). After wp2 because the new SessionStart advisory must follow 011's no-write SessionStart rule and 012's switch semantics. +- wp2: hook runtime fixes (010 overview; 011_issue255_codexclaw_gitignore.md, #255 partial fix; 012 #252 PABCD switch, 013 #253 IDLE goal release, 014 #254 per-turn Stop budget, 015 #251 worker gate, 016 #250 trigger narrowing). Foundation first: 011 adds the shared first-directory helper without changing SessionStart state creation; 012 adds the switch every later PABCD handler consults; 013 and 014 then change Stop continuation; 015 and 016 are leaf policy changes. Built in parallel by gpt-6-sol builders on separate branches in task-owned worktrees, merged in that order. +- wp3: agent-created thread permissions (020 hook and advisory, 021 dispatch guidance plus #265 checkpoint guidance). After wp2 so the read-only advisory can follow 011's directory contract and 012's switch semantics. - wp4: goalplan pending decisions (030, #262). Independent of wp3; after wp2 so Stop/IDLE logic changes are settled before readiness semantics change. - wp5: delivery and issue disposition (040). @@ -50,10 +50,9 @@ Delivery: one ordinary PR per implementation work phase from a `codex/issue-trai | #252 | implement (narrowed) | 012 | | #253 | fix | 013 | | #254 | implement | 014 | -| #255 | fix | 011 | +| #255 | partial fix (gitignore); lazy creation deferred because stateless sessions affect evidence/goal gates and atomic publication | 011_issue255_codexclaw_gitignore.md | | #262 | implement | 030 | | #265 | guidance only | 021 | | agent-thread permissions (no issue) | implement | 020, 021 | | #209 #213 #247 #256 #257 #258 #259 #260 #263 #264 #266 #267 #268 | defer | 001 | | #261 | decline | 001 | - diff --git a/devlog/_plan/260927_issue_train/001_research.md b/devlog/_plan/260927_issue_train/001_research.md index 13ec79e0..2f9b1d07 100644 --- a/devlog/_plan/260927_issue_train/001_research.md +++ b/devlog/_plan/260927_issue_train/001_research.md @@ -11,8 +11,8 @@ Six gpt-6-sol explorers verified each issue against `dev` at `958441a9` (read-on | #251 SubagentStop gate blocks built-in worker | REAL | fix (015) | The gate checks agent type without checking whether the parent armed a PABCD cycle (subagent-evidence.ts:471). | | #252 no supported PABCD off switch | PROPOSAL | implement narrowed (012) | A blanket pabcd-state no-op would also remove worktree, memory-write and automation guards (inventory.json:175-205); a PABCD-policy switch keeps them. | | #253 GOAL-IDLE-CONTINUE-01 blocks every active-goal session | REAL | fix (013) | handleStop blocks at IDLE whenever a native goal is active, even with no state or goalplan (hook.ts:1781-1787; hook-continuation.test.ts:506 pins it). | -| #254 MAX_STOP_BLOCKS_TOTAL never resets | PROPOSAL (behavior intended) | implement (014) | The cap is documented as per-session (hook.ts:1365); long goal sessions still lose continuation silently. A per-real-user-turn budget keeps the unattended bound. | -| #255 SessionStart writes session state into every cwd | REAL | fix (011) | handleSessionStart calls ensureState unconditionally (hook.ts:572, state.ts:316-368). | +| #254 MAX_STOP_BLOCKS_TOTAL never resets | PROPOSAL (behavior intended) | implement (014) | The cap is documented as per-session (hook.ts:1371); long goal sessions still lose continuation silently. A per-real-user-turn budget keeps the unattended bound. | +| #255 SessionStart writes session state into every cwd | REAL | partial fix: `.codexclaw/.gitignore` on first directory creation (011_issue255_codexclaw_gitignore.md); lazy creation deferred | `handleSessionStart` calls `ensureState` unconditionally (`hook.ts:572-575`, `state.ts:368-381`). The issue is low severity and offers an ignore-file alternative. Audit round 3 showed that stateless sessions affect executor evidence and goal-complete gates (`goal-gate.ts:215-236`) and conditional atomic publication (`state.ts:607-617`); lazy creation needs its own design. | | #256 independent verification receipt | PROPOSAL | defer | Receipts store a joined command string with no argv, cwd or output digests (receipt-cli.ts:170-172); a rerun cannot be reconstructed reliably yet. | | #257 strict accepted-progress report | PROPOSAL | defer | Review rounds attach to plan audit, not completed work (review-round-cli.ts:230-235); needs a post-implementation acceptance record first (#256). | | #258 verbatim archive of user-typed prompts | PROPOSAL | defer | UserPromptSubmit carries no typed-versus-injected provenance (hook.ts:134, parse.ts:65); the archive cannot promise what it claims. | @@ -27,9 +27,8 @@ Six gpt-6-sol explorers verified each issue against `dev` at `958441a9` (read-on | #267 suggested wave width | PROPOSAL | defer | Family, native limit and open-child count are not observable at SessionStart (dispatch-card.ts:24-36). | | #268 compact-at-clean-boundary advisory | PROPOSAL | defer | No token usage parsing exists and Stop systemMessage display is unverified (stop.rs:288). | -Declined and deferred issues stay open with a comment linking this record, except #261, which is closed as declined. +Declined and deferred issues stay open with a comment linking this record, except #261, which is closed as declined. #255 also stays open after its partial `.gitignore` fix, with the PR linked and lazy creation deferred. ## Agent-created thread permissions (no issue) Measured on this Mac on 2026-09-27. Rollouts with `thread_source=agent_created_thread` started with `approval_policy=on-request` and `workspace-write` in 7 of about 356 September cases (plus 8 archived on 09-23), including a projectless thread on 09-27, while their parents ran `never` with `danger-full-access`. The Desktop per-thread store held 5 `:workspace` entries out of 1,159. The only command-approval responses in the retained Desktop logs (05:06 and 08:53 UTC on 09-27) came from such children; thread `01a0e145` contains the exact `git fetch origin dev --quiet` command approved at 08:53. Subagents followed their parent in every case (5,336 never to never, 40 on-request to on-request). Codex runs PermissionRequest hooks before the user approval UI (codex-rs/core/src/tools/approvals.rs:505-525), and the bundled runtime 0.158.0-alpha.2.1 contains that path. Upstream: openai/codex #33282, #40793, #41167. - diff --git a/devlog/_plan/260927_issue_train/002_architect_consultation.md b/devlog/_plan/260927_issue_train/002_architect_consultation.md index df25465f..4aa073ce 100644 --- a/devlog/_plan/260927_issue_train/002_architect_consultation.md +++ b/devlog/_plan/260927_issue_train/002_architect_consultation.md @@ -10,7 +10,7 @@ Handle: `01a0e32b-c89d-7901-8a08-4af390a8081d` (gpt-6-sol, CXC-ROLE: architect, | AD-2 | Opt-in `permissions.agentCreatedThreadAutoAllow` in user-global `$CODEXCLAW_HOME/config.json`; project files cannot enable it | Accepted | | AD-3 | Eligibility from the first bounded line of `transcript_path` (session_meta id match, thread_source agent_created_thread), `permission_mode=default`, no agent fields, top-level never plus danger-full-access in CODEX_HOME config, any profile means unknown | Accepted; forked threads stay out until measured | | AD-4 | Matcher `*`, self-filter Bash, write_stdin, apply_patch, request_permissions and `mcp__` names; exact allow JSON | Accepted | -| AD-5 | SessionStart advisory regardless of opt-in, claims only degraded approval mode | Accepted; must obey 011's no-write SessionStart rule | +| AD-5 | SessionStart advisory regardless of opt-in, claims only degraded approval mode | Accepted; advisory itself stays read-only under 011's directory-creation rule | | AD-6 | Guidance: create_worktree plus a subagent that passes the worktree as shell workdir; threads stay the surface for lanes needing their own goal | Accepted | | AD-7 | Bypass record: C4, PermissionRequest hook, opt-in allow suppresses user approval, residual risks listed | Accepted | @@ -31,7 +31,13 @@ Upstream gap: complete coverage needs Codex to expose the resolved sandbox state Verdict: MISALIGNED on AD-3 wording and AD-4 MCP scope; AD-1, AD-2, AD-5, AD-6, AD-7 ALIGNED. Gaps and main dispositions: 1. The success claim said the hook proves full access; it only sees top-level config evidence and a tool-name convention. Folded: 020's success line now says "explicit top-level config evidence" and names the `mcp____` convention, and states the exact guarantee needs upstream fields. -2. Cross-phase invariants were unpinned. Folded: both permission verbs stay active under `CODEXCLAW_PABCD=off` (020 dependency line, 012 note), and 020 requires a built-CLI integration test with the switch off and a fresh cwd that asserts output and no `/.codexclaw`. The CLI hook-observation write goes to `CODEX_HOME`, which 011's rule allows. +2. Cross-phase invariants were unpinned. Folded: both permission verbs stay active under `CODEXCLAW_PABCD=off` (020 dependency line, 012 note), and 020 requires a built-CLI integration test with the switch off and a fresh cwd that asserts the advisory produces output without creating `/.codexclaw` when called alone. When PABCD policy is enabled, the separate generic SessionStart hook still creates state and the ignore file. The CLI hook-observation write goes to `CODEX_HOME`, which 011's rule allows. 3. Build and inventory order. Folded: build first, then focused tests, then full `npm test`, then `inventory.mjs --write/--check --tests ` and gate. No further module-ownership conflicts were found across 010-016, 020 and 030. + +## Re-plan after audit round 3 + +Three audit rounds returned **FAIL**. The root cause was the proposed lazy SessionStart state creation: stateless sessions would change the executor evidence gate and the goal-complete gate (`plugins/codexclaw/components/pabcd-state/src/goal-gate.ts:215-236`), and the proposed `writeExistingState` guard did not settle atomic conditional publication against `writeState`'s temporary-file rename (`state.ts:607-617`). The earlier no-parent-state executor release was especially unsafe. + +Main changed #255 to a low-severity **partial fix** using the reporter's offered `.codexclaw/.gitignore` alternative. SessionStart and CLI identity behavior stay as shipped; `011_issue255_codexclaw_gitignore.md` now routes first-directory writers through one exclusive ignore-file helper. Lazy state creation, `verifiedCreateState`, `sessionStateFileExists`, `writeExistingState`, missing-state hook suppression, and no-parent-state evidence release were dropped from this train and deferred for a separate design. The revised 010/013/014/015/020 contracts follow that decision. diff --git a/devlog/_plan/260927_issue_train/010_wp2_runtime_overview.md b/devlog/_plan/260927_issue_train/010_wp2_runtime_overview.md index ee6e7390..ec326fc1 100644 --- a/devlog/_plan/260927_issue_train/010_wp2_runtime_overview.md +++ b/devlog/_plan/260927_issue_train/010_wp2_runtime_overview.md @@ -1,10 +1,10 @@ # wp2 — Runtime issue train (#255, #252, #253, #254, #251, #250) -This work phase makes idle Codex sessions passive, provides a PABCD off switch, and narrows continuation and delegation gates to the sessions they govern. Six independent builder branches should land in this order: `011` lazy state, `012` switch, `013` idle release, `014` turn budget, `015` worker gate, `016` trigger narrowing. Rebase each later branch on the just merged predecessor, rerun its targeted tests, then run the complete phase gate. All anchors below refer to branch `codex/issue-train-0927` at `958441a9`; each builder must recheck them after predecessor merges. +This work phase reduces false continuation and delegation gates, adds a PABCD off switch, and gives newly created `.codexclaw` directories a local-state ignore file. Six independent builder branches should land in this order: `011` gitignore, `012` switch, `013` idle release, `014` turn budget, `015` worker gate, `016` trigger narrowing. Rebase each later branch on the just merged predecessor, rerun its targeted tests, then run the complete phase gate. All anchors below refer to branch `codex/issue-train-0927` at `958441a9`; each builder must recheck them after predecessor merges. ## Phase contract -- `011` owns first-write identity and the SessionStart audit. It must land first because every later hook can encounter a missing state file. +- `011` owns the first-directory `.gitignore` helper. SessionStart still creates the default state file; lazy state creation is deferred. - `012` owns the off switch in hook dispatch. The safety guards remain active; see its explicit allowlist. - `013` makes active but unbound goals release at IDLE without writing counters. - `014` makes the absolute Stop cap apply to one real user turn. Native Stop continuations do not create new UserPromptSubmit inputs: `/tmp/cxc-perm/codex-src/codex-rs/core/src/session/turn.rs:666-683`, `/tmp/cxc-perm/codex-src/codex-rs/core/src/hook_runtime.rs:682-707`. @@ -15,9 +15,9 @@ This work phase makes idle Codex sessions passive, provides a PABCD off switch, | Path | Planned owners | Conflict resolution | | --- | --- | --- | -| `plugins/codexclaw/components/pabcd-state/src/hook.ts` | 011, 013, 014, 016 | Preserve 011's missing-state behavior; apply 013 around `handleStop` guard 2a, 014 around `handleUserPromptSubmit` and `bumpStopCounter`, then 016 detector replacements. | -| `plugins/codexclaw/components/pabcd-state/src/cli.ts` | 012, possibly 011 | Keep 012 dispatch guard above the PABCD-only branches while preserving 011's mutating CLI identity checks. | -| `plugins/codexclaw/components/pabcd-state/src/state.ts` | 011, 014 | Keep `ensureState` exclusive-create contract and add 014's `stopBlockTurnId` to `State`, default, and strict reconstruction. | +| `plugins/codexclaw/components/pabcd-state/src/hook.ts` | 013, 014, 016 | Keep SessionStart calling `ensureState`; apply 013 around `handleStop` guard 2a, 014 around `handleUserPromptSubmit` and `bumpStopCounter`, then 016 detector replacements. | +| `plugins/codexclaw/components/pabcd-state/src/cli.ts` | 012 | Keep 012 dispatch guard above the PABCD-only branches; 011 makes no CLI identity change. | +| `plugins/codexclaw/components/pabcd-state/src/state.ts` | 011, 014 | Add `ensureCodexclawDir` before existing child creation, keep `ensureState` exclusive-create contract, and add 014's `stopBlockTurnId` to `State`, default, and strict reconstruction. | | `plugins/codexclaw/components/pabcd-state/test/hook*.test.ts` | 011–016 | Merge by named tests; update existing expectations instead of retaining contradictory tests. | No builder may overwrite another branch's full file. Resolve conflicts in the listed order; run the named tests after each merge. The PABCD hook dispatch is `cli.ts:329-488`; state serialization is `state.ts:486-615`; test globs are in root `package.json:24`. @@ -37,8 +37,8 @@ The builders run `node --test ` after each fix, then ` | Scenario | Expected | | --- | --- | -| Fresh `SessionStart`, no `.codexclaw` | No directory or state file created (`011`). | -| First verified mutating command | Exclusive state creation; failed native verification leaves no state (`011`). | +| Fresh `SessionStart`, PABCD enabled, no `.codexclaw` | Creates default session state and the exact `.codexclaw/.gitignore` (`011`). | +| Any other first `.codexclaw` writer | Creates the directory through `ensureCodexclawDir` and publishes `.gitignore` exclusively (`011`). | | `CODEXCLAW_PABCD=off` or project `pabcd.enabled=false` | PABCD hooks silent; independent safety guards still run (`012`). | | Active host goal, no bound plan, IDLE Stop | Release without counter write (`013`). | | Bound goal, IDLE Stop | Existing bounded arming behavior (`013`). | @@ -48,4 +48,4 @@ The builders run `node --test ` after each fix, then ` ## Out of scope -No relocation to `CODEXCLAW_HOME`, blanket `.gitignore`, change to explicit `orchestrate` parser grammar, new hook registration, or relaxation of worktree/memory/automation safety gates. No source or test change is made in this planning pass. +No relocation to `CODEXCLAW_HOME`, lazy state creation, change to explicit `orchestrate` parser grammar, new hook registration, or relaxation of worktree/memory/automation safety gates. No source or test change is made in this planning pass. diff --git a/devlog/_plan/260927_issue_train/011_issue255_codexclaw_gitignore.md b/devlog/_plan/260927_issue_train/011_issue255_codexclaw_gitignore.md new file mode 100644 index 00000000..0aefd962 --- /dev/null +++ b/devlog/_plan/260927_issue_train/011_issue255_codexclaw_gitignore.md @@ -0,0 +1,51 @@ +# #255 — Ignore local runtime state when CodexClaw creates `.codexclaw` + +**Partial fix.** Keep `handleSessionStart` and `ensureState` exactly as they behave today: a fresh SessionStart with PABCD policy enabled creates `.codexclaw/sessions/.json`. Do not introduce lazy state creation, `verifiedCreateState`, `sessionStateFileExists`, `writeExistingState`, or CLI identity changes. When CodexClaw itself first creates `/.codexclaw`, create `/.codexclaw/.gitignore` exclusively. This addresses the reporter's offered ignore-file alternative without changing session or gate semantics (issue #255, severity low). + +Lazy creation is **DEFERRED** to its own design. The round-3 audit found that absent session state changes the executor evidence gate and goal-complete gate (`plugins/codexclaw/components/pabcd-state/src/goal-gate.ts:215-236`), while guarded writes would have to preserve `writeState`'s atomic conditional publication (`plugins/codexclaw/components/pabcd-state/src/state.ts:607-617`). This phase does not make those changes. + +## File change map + +### ADD helper in `plugins/codexclaw/components/pabcd-state/src/state.ts` + +Export `ensureCodexclawDir(cwd): string`. Attempt `mkdirSync(join(cwd, STATE_DIR))` **without** `recursive`; only the caller that successfully creates that exact directory writes `.gitignore` with `writeFileSync(path, GITIGNORE_TEXT, { flag: "wx" })`. Ignore `EEXIST` from the directory creation and from the exclusive file creation; propagate other errors. An existing `.codexclaw` directory is never modified, even if it lacks `.gitignore`; an existing `.gitignore` is never overwritten. Do not use an `existsSync` precheck to decide ownership. Concurrent callers may race; one creates the directory and at most one publishes the ignore file. Keep `ensureState`'s exclusive session-file publication unchanged (`state.ts:359-410`). + +Use these exact bytes (including the final newline): + +```gitignore +# CodexClaw wrote this when it created .codexclaw; everything here is local runtime state. +* +!.gitignore +!rules/ +!rules/*.md +``` + +`rules/*.md` is the one deliberate exception: `rules.ts:4-6,20-29` reads user-authored project rules there, so they must remain eligible for a commit. The directory re-include makes Git traverse it and the `*.md` re-include exposes only the files the reader accepts. Goalplans are generated and mutated local loop state (`goalplan.ts:13,850-888`; `skills/loop/references/durable-goalplan.md:45-48`), so do not re-include `goalplans/` or its ledger. Other sessions, evidence, ledgers, and local configs stay ignored. + +### ROUTE all project-local first creators through the helper + +Call `ensureCodexclawDir(cwd)` before any recursive child-directory creation. Keep the existing child `mkdirSync` and publication behavior. The `rg -n 'mkdirSync' plugins/codexclaw/components/*/src` inventory gives these project-local writers: + +| Owner | First-create path | Action | +| --- | --- | --- | +| `pabcd-state/src/state.ts` | `ensureState:377`, `writeState:609`, lock:656, ledger:683, interview ledger:740 | Call helper before each child creation; SessionStart still calls `ensureState` (`hook.ts:572-575`). | +| `pabcd-state/src/{friction,render-observations,edit-shape}.ts` | `friction:127`, `render-observations:72,127`, `edit-shape:166` | Replace root `mkdirSync` with helper. | +| `pabcd-state/src/{metrics,divergence,interview-ledger,subagent-evidence}.ts` | `metrics:132,138`, `divergence:123,194`, `interview-ledger:220`, `subagent-evidence:217,338,353` | Call helper before child creation; no evidence-gate behavior change. | +| `pabcd-state/src/{receipt-cli,session-source,goalplan,freeze-cli,release-cli,worktree-guard}.ts` | `receipt-cli:181`, `session-source:183`, `goalplan:853,877` (lock `:754` needs an existing plan), `freeze-cli:118`, `release-cli:135`, `worktree-guard:527` | Call helper before project-local child creation; keep existing file/lock rules. | +| `pabcd-state/src/rule-impact-ledger.ts` | `:101-106` takes an arbitrary output path | When the resolved path is under the target cwd's `.codexclaw`, call helper first; thread cwd through the caller if needed. Arbitrary output paths retain their existing behavior. | +| `cxc-ops/src/{map-affordance,activation-trace}.ts` | recovery root `map-affordance:241-249`; trace `activation-trace:130-139` | Use helper when creating the project-local root; preserve recovery/trace best-effort behavior. `scouting-bundle.ts:126` only reads sessions. | +| `bg-wake/src/store.ts` | `ensureDir:45-48` | Call helper before `bg/`; this can be the first writer even without SessionStart. Orphan adoption itself writes only after finding an existing record (`registry.ts:236-249`). | +| `subagent-config/src/store.ts` | `writeRaw:251-259` via project `storePath` | Route project-local `subagents.json` writes through helper; do not apply it to a user-global home path. | +| `subagent-config/src/fallback-dispatch.ts` | `directory:70-78` creates `.codexclaw/dispatches/` | Replace only the root `.codexclaw` creation with the helper; retain its existing symlink and child-directory checks. The lock-directory calls at `:137,256` have `directory()` as a prerequisite and cannot be first. | +| `messenger-bridge/src/{db,event-log}.ts` | `db:1038-1039` and `bridge-controller:93` → `event-log:55` | Route these cwd-local writes through helper. `service.ts` and `token-intake.ts` mkdir calls target home/service locations; `event-log` can accept an explicit project cwd or have its caller ensure the root. | + +The other `mkdirSync` results do not first-create `/.codexclaw`: `plan-cli.ts:178-184` writes `devlog/_plan`; `recall/src/index-db.ts:20-22,90-94` and `skill-search/src/cache.ts:16,49` write the user-global cache; `config-guard`, `subagent-config/src/role-registration.ts:47-67`, and `subagent-config/src/spawn-attach-hook.ts:365-380` target `CODEX_HOME` or a temporary directory; `messenger-bridge/src/service.ts` writes its home service directory. `release-cli.ts` can also accept an arbitrary `--candidate` path: call the helper only for the default project-local release path, preserving arbitrary output behavior. `rule-impact-ledger.ts` likewise accepts an arbitrary path. Recheck each path and import/build boundary in the implementation branch before changing it. If another cwd-local first writer appears, route it through this same helper. + +### MODIFY tests + +- `pabcd-state/test/state.test.ts`: fresh-cwd `handleSessionStart` returns `""` and creates both the default session file and `.codexclaw/.gitignore` with the **exact bytes** above. Keep the existing `ensureState` default/exclusive-create assertions (`state.test.ts:27-68`). +- Existing `.codexclaw` without `.gitignore` stays without it after `ensureState`; an existing `.gitignore` keeps its exact bytes. Concurrent first creators tolerate `EEXIST`; after both finish, one ignore file has exact content and session state remains valid. +- In a temporary Git repository, `git check-ignore` reports `sessions/.json` and the ledger ignored, while `.gitignore` and `rules/example.md` are not ignored. Test a `goalplans//goalplan.json` path as ignored too. Use `git check-ignore -q` exit statuses, not an ungrounded visual inspection. +- Add focused first-writer tests for representative independent components (at least `cxc-ops` PostCompact recovery, bg-wake `ensureDir`, and a non-state PABCD writer); assert the same ignore file. No source/test edit occurs in this docs-only phase. + +**Verification for the implementation branch:** focused `node --test` on touched component test files, `npm run build`, `npm test`, `npm run gate`, and the inventory check named in 010. Re-run `rg -n 'mkdirSync' plugins/codexclaw/components/*/src` and account for every project-local writer after the patch. The fix is partial because the directory and session file still appear at SessionStart; no promise of a read-only session source is made. diff --git a/devlog/_plan/260927_issue_train/011_issue255_lazy_session_state.md b/devlog/_plan/260927_issue_train/011_issue255_lazy_session_state.md deleted file mode 100644 index 17b51110..00000000 --- a/devlog/_plan/260927_issue_train/011_issue255_lazy_session_state.md +++ /dev/null @@ -1,106 +0,0 @@ -# #255 — Create session state on the first verified mutation - -`SessionStart` must leave a new repository without `.codexclaw`. The first `cxc session bind`, `cxc orchestrate --session `, or `cxc loop init --session ` creates the state after the same native identity check. Reads and failed verification remain read-only. - -Current anchors: `plugins/codexclaw/components/pabcd-state/src/hook.ts:572`, `plugins/codexclaw/components/pabcd-state/src/state.ts:368`, `plugins/codexclaw/components/pabcd-state/src/session-cli.ts:76`, `plugins/codexclaw/components/pabcd-state/src/orchestrate-cli.ts:538`, `plugins/codexclaw/components/pabcd-state/src/goalplan-cli.ts:675`. - -## File change map - -### MODIFY `plugins/codexclaw/components/pabcd-state/src/hook.ts` - -At `handleSessionStart` (`:567-576`), replace `ensureState(payload.cwd, payload.session_id)` with a read-only existing-file check, or simply remove it: the handler has no refresh work today. Keep its empty output. `ensureState` at `state.ts:368-410` creates `.codexclaw/sessions`; `readState` supplies a virtual default without writing (`state.ts:486-603`). Do not call `writeState` from SessionStart. If the hook needs to refresh an existing file in a later change, check existence and preserve its contents. - -```diff - export function handleSessionStart(payload: SessionStartPayload): string { - if (payload.hook_event_name !== "SessionStart") return ""; -- ensureState(payload.cwd, payload.session_id); - return ""; - } -``` - -Also protect ordinary UserPromptSubmit from minting state: `hook.ts:643-653` currently writes a memory-request marker before it knows whether a state file exists, and `:698-705` writes `loopArmSeen`. At `:630`, compute `const stateExists = sessionStateFileExists(payload.cwd, payload.session_id)` using the new read-only helper below. Guard each `writeState` in this handler with `stateExists`; on a missing file, allow context-only guidance but never persist `injectedTurns`, `loopArmSeen`, or the memory marker. A memory request still has to be enforced separately by the independent PreToolUse memory gate; the no-state case must deny rather than silently authorize. Add a test that a fresh normal prompt and a fresh loop-arm prompt leave `.codexclaw` absent. An explicit chat `orchestrate` command is a mutating hook path at `hook.ts:659-662`; on missing state it must emit a “run cxc session bind or verified cxc orchestrate” instruction and avoid calling `handleOrchestrateCommand`, which otherwise writes an unverified state. - -### MODIFY `plugins/codexclaw/components/pabcd-state/src/state.ts` - -Expose a read-only existence helper beside private `statePath` (`:320-322`); `existsSync` is already imported. This prevents callers from accidentally defaulting a missing file into a write. - -```ts -export function sessionStateFileExists(cwd: string, sessionId: string): boolean { - return isCanonicalSessionId(sessionId) && existsSync(statePath(cwd, sessionId)); -} -``` - -Do not change the exclusive-create `ensureState` algorithm (`:368-410`). - -### MODIFY `plugins/codexclaw/components/pabcd-state/src/orchestrate-cli.ts` - -`runOrchestrateCli` requires explicit `--session` (`:519-535`) and currently rejects a missing file (`:538-548`). Replace only that missing-file branch. Reuse `resolveNativeSession(args.cwd, nativeEnv)` from `session-binding.ts:41-109`; require `native.ok`, exact `native.sessionId === sessionId`, and exact native cwd. Then call `ensureState(native.cwd, native.sessionId)` and inspect/read it. A reserved standalone `cli` key keeps its existing behavior. A mismatch, absent `CODEX_THREAD_ID`, missing/archived/subagent native row, wrong cwd, or malformed existing state fails before any mutation. `session-cli.ts:76-91` is the reference for verify → exclusive create → inspect. Status at `orchestrate-cli.ts:499-517` remains read-only. - -Replace `orchestrate-cli.ts:543-548` with this complete block: - -```ts -if (args.session && !sessionFileExists(args.cwd, sessionId) && !RESERVED_SESSION_KEYS.has(sessionId)) { - const native = resolveNativeSession(args.cwd, nativeEnv); - if (!native.ok || native.sessionId !== sessionId) { - return { code: 1, output: `orchestrate ${args.verb}: native session verification failed; nothing was written` }; - } - try { ensureState(native.cwd, native.sessionId); } - catch { return { code: 1, output: `orchestrate ${args.verb}: could not create session state; nothing was written` }; } - const checked = inspectState(native.cwd, native.sessionId); - if (!checked.ok || !checked.stateExists) { - return { code: 1, output: `orchestrate ${args.verb}: session state invalid after creation; nothing was written` }; - } -} -``` - -`inspectState` is currently private (`session-cli.ts:13`); export it and import it into `orchestrate-cli.ts`, along with `ensureState`. This reuses the raw regular-file/symlink and session-id checks that `session bind` performs. Do not invent a second native verifier. Preserve `resolveSessionSource` before phase writes (`orchestrate-cli.ts:568-572`). - -### MODIFY `plugins/codexclaw/components/pabcd-state/src/session-cli.ts` - -Change `function inspectState` at `:13` to `export function inspectState` without changing its body. `session bind` already calls it before and after exclusive creation (`:76-91`). - -### MODIFY `plugins/codexclaw/components/pabcd-state/src/goalplan-cli.ts` - -The bound `init` path checks source identity at `:669-679`, then writes the plan and finally writes default session state at `:681-699`. Move native verification plus exclusive state creation before `writeGoalplan` so rejected identities leave neither plan nor state. Use the same `resolveNativeSession` contract and exact session/cwd match; skip this step for unbound `init` to preserve its local-artifact behavior (`:672-674`). After creation, preserve source identity checking and the existing slug binding. The `runGoalplanCli` signature needs injectable `nativeEnv: NodeJS.ProcessEnv = process.env` for focused tests, or a shared verified-create helper that both CLIs call. A helper belongs in `session-binding.ts`, which already owns native verification; no copy of SQLite logic. - -Change the `runGoalplanCli` signature at `:651` to `runGoalplanCli(args: GoalplanCliArgs, nativeEnv: NodeJS.ProcessEnv = process.env)`. Replace the bound-session check block at `:675-680` with this complete block, leaving `writeGoalplan` at `:686` and slug binding at `:695-698` in place: - -```ts -if (typeof args.session === "string" && args.session.length > 0) { - const native = resolveNativeSession(args.cwd, nativeEnv); - if (!native.ok || native.sessionId !== args.session) { - return { output: "loop init: native session verification failed; nothing was written", code: 1 }; - } - const before = inspectState(native.cwd, native.sessionId); - if (!before.ok) return { output: `loop init: ${before.error}; nothing was written`, code: 1 }; - const gate = checkBoundSourceIdentity(args.cwd, args.session); - if (!gate.ok) return { output: `loop init: ${gate.reason}\nNothing was written.`, code: 1 }; - try { ensureState(native.cwd, native.sessionId); } - catch { return { output: "loop init: could not create session state; nothing was written", code: 1 }; } - const after = inspectState(native.cwd, native.sessionId); - if (!after.ok || !after.stateExists) { - return { output: "loop init: state invalid after creation; nothing was written", code: 1 }; - } -} -``` - -Import `resolveNativeSession` from `session-binding.ts`, `inspectState` from `session-cli.ts`, and `ensureState` from `state.ts` at `goalplan-cli.ts:40-48`. Existing `cli.ts:171` passes no second argument and therefore uses the live environment. Any synthetic-ID `loop init --session` tests must create a native fixture or expect rejection; only unbound init retains its old no-native path. - -### READ ONLY `plugins/codexclaw/components/cxc-ops/src/map-affordance.ts` - -`runMapAffordanceSessionStart` is context-only (`:296-337`) and does not call `recoveryPath`. `recoveryPath` can create `.codexclaw/affordance-recovery` at `:241-249` only for `runPostCompactAffordance` (`:257-265`), not SessionStart. Keep this distinction and add a regression check; no production diff needed unless the branch discovers another SessionStart call to `recoveryPath(..., true)`. - -### No production change: recall and bg-wake - -Recall SessionStart builds context at `recall/src/hook.ts:699-737`; its index writes are under the user's home, not the cwd `.codexclaw` (`recall/src/index-db.ts:5-22,93`). `bg-wake/src/hook.ts:114-127` calls orphan adoption. `bg-wake/src/registry.ts:236-249` only writes when `listRecords` finds an existing terminal undelivered record, so a cwd with no `.codexclaw` creates nothing; retain adoption because it preserves completed background work across restart. Assert empty-directory behavior in tests. `cxc-ops` PostCompact marker is a separate event and may still create a recovery directory (`map-affordance.ts:257-265`). - -### MODIFY tests - -- `plugins/codexclaw/components/pabcd-state/test/state.test.ts:27-68`: rename the test to `ensureState: first verified mutation creates exact default state`; keep its exact default/temporary-file assertions. Add `SessionStart: fresh cwd remains without .codexclaw`, call `handleSessionStart`, assert empty output and `existsSync(join(cwd, STATE_DIR)) === false`; this fails before the fix. -- `plugins/codexclaw/components/pabcd-state/test/orchestrate-cli.test.ts`: `missing native state is created by matching verified orchestrate mutation`; `unknown explicit id or wrong cwd leaves .codexclaw absent`; `status missing state stays read-only`. Each must inspect exact state file and exit code. Existing native fixture helpers should be reused. -- `plugins/codexclaw/components/pabcd-state/test/goalplan.test.ts`: `bound loop init verifies native identity before plan/state writes`; assert wrong ID leaves both paths absent and matching ID creates both. This fails today because the bound path writes a default state without native verification. -- `plugins/codexclaw/components/cxc-ops/test/map-affordance.test.ts`, `plugins/codexclaw/components/bg-wake/test/hook.test.ts`: call SessionStart in an empty cwd and assert no `.codexclaw` exists. - -## Activation and bypass record - -Fresh SessionStart, resumed SessionStart with an existing state, fresh prompt, first valid mutation, invalid native ID/cwd/source, two concurrent creators, and empty bg/recall/map hooks must each be exercised. Native verification tier: CLI boundary; executing surface: `session bind`, orchestrate, bound loop init; known bypass: a direct library caller can still call `writeState`, and hook payloads are not independently authenticated; residual risk: same-user tampering with the native DB/env; wording: “verified CLI first mutation,” not a universal security boundary. Final enforcement layer is each mutating CLI entry. No relocation of `.codexclaw`, blanket ignore rule, or new SessionStart write. diff --git a/devlog/_plan/260927_issue_train/012_issue252_pabcd_switch.md b/devlog/_plan/260927_issue_train/012_issue252_pabcd_switch.md index 49177917..5512b2c2 100644 --- a/devlog/_plan/260927_issue_train/012_issue252_pabcd_switch.md +++ b/devlog/_plan/260927_issue_train/012_issue252_pabcd_switch.md @@ -1,6 +1,6 @@ # #252 — PABCD hook policy switch -`CODEXCLAW_PABCD=off` disables PABCD hook behavior for the process. Otherwise a project root `codexclaw.json` with `{ "pabcd": { "enabled": false } }` disables it. Missing, malformed, or other values default to enabled. The environment setting wins over the project setting. This changes hook dispatch only; it does not erase state or disable CLI commands. +`CODEXCLAW_PABCD` is a two-way override: normalized `off|0|false` disables and `on|1|true` enables PABCD hooks, regardless of project config. An unrecognized value is ignored; then project-root `codexclaw.json` `{ "pabcd": { "enabled": false } }` disables. Missing or malformed project config defaults to enabled. This changes hook dispatch only; it does not erase state or disable CLI commands. Current anchors: `plugins/codexclaw/components/pabcd-state/src/cli.ts:329`, `plugins/codexclaw/components/pabcd-state/src/interview-policy.ts:26`, `plugins/codexclaw/components/pabcd-state/src/goal-gate.ts:313`, `docs-site/src/content/docs/guides/pabcd.md:67`. @@ -8,11 +8,13 @@ Current anchors: `plugins/codexclaw/components/pabcd-state/src/cli.ts:329`, `plu ### MODIFY `plugins/codexclaw/components/pabcd-state/src/interview-policy.ts` -This module already owns the project config filename and fail-safe JSON read (`:26-61`). Add `readPabcdEnabled(cwd: string, env: NodeJS.ProcessEnv = process.env): boolean` beside `readInterviewPolicy`. Use `configPath(cwd)`; parse only a plain object with a plain-object `pabcd` member and boolean `enabled`. The only disabling values are exact env `off` (after `trim().toLowerCase()`) and exact JSON boolean `false`. No write is needed: `writeInterviewPolicy` preserves unrelated keys (`:63-95`). +This module already owns the project config filename and fail-safe JSON read (`:26-61`). Add `readPabcdEnabled(cwd: string, env: NodeJS.ProcessEnv = process.env): boolean` beside `readInterviewPolicy`. Normalize the env value with `trim().toLowerCase()` and handle the recognized enable/disable sets **before** reading project config. For an unrecognized env value, use `configPath(cwd)`; parse only a plain object with a plain-object `pabcd` member and boolean `enabled`. Only exact JSON boolean `false` disables at that layer. No write is needed: `writeInterviewPolicy` preserves unrelated keys (`:63-95`). ```ts export function readPabcdEnabled(cwd: string, env: NodeJS.ProcessEnv = process.env): boolean { - if (env.CODEXCLAW_PABCD?.trim().toLowerCase() === "off") return false; + const override = env.CODEXCLAW_PABCD?.trim().toLowerCase(); + if (override === "off" || override === "0" || override === "false") return false; + if (override === "on" || override === "1" || override === "true") return true; try { const raw: unknown = JSON.parse(readFileSync(configPath(cwd), "utf8")); if (!raw || typeof raw !== "object" || Array.isArray(raw)) return true; @@ -55,7 +57,7 @@ Import `readPabcdEnabled` from `interview-policy.ts` at `cli.ts:23-53`. Use `pab Split mixed branches, rather than skipping independent logic: `pre-tool-use-edit :438-442` must still execute `handleApplyPatchLint` but must skip `handleIdleEditAdvisory`; `post-tool-use-edit-shape :459-467` can no-op because both its shape hint and render capture serve PABCD; `pre-tool-use-lint :433-436` always stays. `session-start-rules :478-480` and `worktree-guard :451-454` stay, because project rules and worktree identity are not PABCD policy. `pre-tool-use :395-402` stays for goal-complete and goal-budget safety, but inspect `goal-gate.ts` branches: only `request_user_input` Interview/goal-mode prohibition is PABCD-specific and should return no intervention under the switch. Goal completion, evidence tombstones and budget protection are independent host-goal safety. Do not change recall's separate component hooks. -The switch must cover `SubagentStop` evidence for both executor and worker when disabled; it does not delete old attempts or tombstones. A standalone `subagent-stop-review` observer is PABCD audit state and no-ops. Existing safety hooks registered outside this component remain untouched. +The switch must cover `SubagentStop` evidence for both executor and worker when disabled; it does not delete old attempts or tombstones. When enabled, the registered executor is always receipt-gated and the built-in worker is gated only in an active B/C cycle (015). When disabled, the gate is silent for **both**, even if a prior cycle was armed. A standalone `subagent-stop-review` observer is PABCD audit state and no-ops. Existing safety hooks registered outside this component remain untouched. ### MODIFY `plugins/codexclaw/components/pabcd-state/src/goal-gate.ts` @@ -69,14 +71,24 @@ Add a short “Disable PABCD hooks” subsection near the existing runtime lifec ### MODIFY tests -- `plugins/codexclaw/components/pabcd-state/test/interview-policy.test.ts`: `pabcd switch: env off overrides project enabled`; `project false disables, malformed and missing enable`; `interview policy write preserves pabcd key`. Assert booleans and preserved JSON. +- `plugins/codexclaw/components/pabcd-state/test/interview-policy.test.ts`: table-test every recognized env spelling in both directions, invalid env fallback, missing/malformed config, and preservation of the `pabcd` key by interview-policy writes. Assert exact expected booleans from the matrix below. - `plugins/codexclaw/test/hook-e2e.test.mjs`: `pabcd off silences UserPromptSubmit, Stop, SessionStart, PostCompact, SubagentStop`; invoke the built hook entry with project config/env, assert stdout empty and no new session/attempt file. This fails before the switch because trigger and Stop hooks still emit/write. - Same e2e file: `pabcd off retains worktree, memory, automation and apply_patch lint guards`; feed each registered event an existing deny fixture and assert its denial survives. `pre-tool-use-edit` must still deny a lint violation and emit no idle advisory. - `plugins/codexclaw/components/pabcd-state/test/goal-gate.test.ts`: `pabcd off allows request_user_input while preserving goal completion and budget denials`. +- `plugins/codexclaw/components/pabcd-state/test/subagent-evidence.test.ts` and built-hook e2e: `pabcd off silences registered executor and built-in worker SubagentStop even with an armed B/C parent`; `pabcd on gates executor when project false`; assert no new attempts or tombstones in the disabled case. + +| Env value | Project `pabcd.enabled` | Expected PABCD hooks | +| --- | --- | --- | +| `off`, `0`, `false` (each, case/space normalized) | true or absent | disabled | +| `on`, `1`, `true` (each, case/space normalized) | false | enabled | +| unrecognized (including empty) | false | disabled (project decides) | +| unrecognized (including empty) | true or absent | enabled (project decides) | +| unset | false | disabled | +| unset | true, absent, or malformed config | enabled | ## Activation and bypass record -Exercise env off with project true, env unset with project false, env `ON` with project false, invalid JSON, missing file, root/subagent payloads, and every mixed event. Tier: local hook dispatch policy, not a host-wide guarantee. Executing surface: `pabcd-state` hook CLI. Known bypass: direct library calls, terminal CLI commands, and an uninstalled/untrusted hook; residual risk: other components may issue independent context. Wording downgrade: “PABCD hooks in this component are silent,” not “CodexClaw is disabled.” Final enforcement layer: `cli.ts` hook dispatch plus `goal-gate.ts`'s Interview branch. Out of scope: recall hooks, state deletion, CLI write blocking, and safety-guard disablement. +Exercise every matrix row, root/subagent payloads, executor/worker SubagentStop and every mixed event. **Tier:** local hook dispatch policy and E8 tests; no E1 host-wide tool denial. **Executing surface:** `pabcd-state` hook CLI. **Known bypass:** direct library calls, terminal CLI commands, and an uninstalled/untrusted hook. **Residual risk:** other components may issue independent context. **Wording downgrade:** “PABCD hooks in this component are silent,” not “CodexClaw is disabled.” **Final enforcement layer:** `cli.ts` hook dispatch plus `goal-gate.ts`'s Interview branch and E8 regression tests. Out of scope: recall hooks, state deletion, CLI write blocking, and safety-guard disablement. ## Note from the roadmap reflection diff --git a/devlog/_plan/260927_issue_train/013_issue253_idle_goal_release.md b/devlog/_plan/260927_issue_train/013_issue253_idle_goal_release.md index b50e309c..6b7d3501 100644 --- a/devlog/_plan/260927_issue_train/013_issue253_idle_goal_release.md +++ b/devlog/_plan/260927_issue_train/013_issue253_idle_goal_release.md @@ -1,6 +1,6 @@ # #253 — Release an unbound active goal at IDLE -An active host goal alone must not make an IDLE PABCD session block Stop. The IDLE arming block applies only when `state.slug` resolves to a bound goalplan. A missing state or empty slug releases without creating state or advancing a counter. +An active host goal alone must not make an IDLE PABCD session block Stop. The IDLE arming block applies only when `state.slug` resolves to a bound goalplan. A missing state or empty slug releases without a counter write. SessionStart normally already created state; a direct Stop call without SessionStart may leave `.codexclaw` absent. Current anchors: `plugins/codexclaw/components/pabcd-state/src/hook.ts:1392`, `plugins/codexclaw/components/pabcd-state/src/hook.ts:1787`, `plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts:506`, `plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts:567`. @@ -24,7 +24,7 @@ Update the stale comments at `hook.ts:1646-1653,1760-1766` so they no longer say ### MODIFY `plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts` -- Replace `GOAL-IDLE-CONTINUE-01: active goal at IDLE blocks with the arming command` at `:506-529` with `GOAL-IDLE-CONTINUE-01: active goal without bound plan releases without state write`. Keep the active host goal fixture, assert `handleStop(...) === ""`, `existsSync(join(cwd, ".codexclaw")) === false`, and a second call remains silent. This fails before the fix because the first call blocks and writes state. +- Replace `GOAL-IDLE-CONTINUE-01: active goal at IDLE blocks with the arming command` at `:506-529` with `GOAL-IDLE-CONTINUE-01: active goal without bound plan releases without state write`. Keep the active host goal fixture, call `handleStop` directly without a preceding SessionStart, assert `handleStop(...) === ""`, `existsSync(join(cwd, ".codexclaw")) === false`, and a second call remains silent. A separate SessionStart-path case should assert the pre-existing session file and `.gitignore` remain unchanged. This fails before the fix because the direct first Stop blocks and writes state. - Existing win32 and bounded tests at `:535-565` currently use unbound state. Bind a real plan/slug before calling Stop; retain their platform and cap assertions. Otherwise they would contradict the new contract. - Keep the bound-plan case at `:567-587` and the bound-empty case at `:589-600`. Add `GOAL-IDLE-CONTINUE-01: stale slug releases without counter write`: write state with a nonexistent slug and `stopBlockTotal: 7`, call Stop, assert empty output and unchanged total. diff --git a/devlog/_plan/260927_issue_train/014_issue254_turn_budget.md b/devlog/_plan/260927_issue_train/014_issue254_turn_budget.md index 82f1d60d..aa71cbe5 100644 --- a/devlog/_plan/260927_issue_train/014_issue254_turn_budget.md +++ b/devlog/_plan/260927_issue_train/014_issue254_turn_budget.md @@ -12,7 +12,7 @@ Codex Stop converts `decision:block` to continuation fragments (`/tmp/cxc-perm/c ### MODIFY `plugins/codexclaw/components/pabcd-state/src/state.ts` -Add `stopBlockTurnId: string | null` next to `stopBlockTotal` in `State` (`:135-137`), default it to `null` beside `stopBlockTotal: 0` (`:299-301`), and reconstruct a nonempty string or `null` beside `:551-554`. `writeState` already serializes the whole state (`:607-615`); no new writer needed. Old files reconstruct `null`, and a missing `turn_id` does not reset an existing budget. +Add `stopBlockTurnId: string | null` next to `stopBlockTotal` in `State` (`:135-137`), default it to `null` beside `stopBlockTotal: 0` (`:299-301`), and reconstruct a nonempty string or `null` beside `:551-554`. Export the existing `statePath` function (`state.ts:320-322`) for the `existsSync` guard below; preserve its `sanitizeKey` path construction. `writeState` already serializes the whole state (`:607-615`); no new writer needed. Old files reconstruct `null`, and a missing `turn_id` does not reset an existing budget. ```diff stopBlockTotal: number; @@ -33,18 +33,18 @@ The numeric branch above is the existing inline validation at `state.ts:551-554` ### MODIFY `plugins/codexclaw/components/pabcd-state/src/hook.ts` -In `handleUserPromptSubmit` at `:629-653`, read state before branch-specific writes. After the existing memory-write marker step (which may update state), re-read state, then reset only when `turn !== ""`, the state file already exists, and `state.stopBlockTurnId !== turn`. Persist `{ ...state, stopBlockTotal: 0, stopBlockTurnId: turn }`, then use this fresh snapshot in the rest of the handler. Do the reset before the `injectedTurns` dedupe, but the persisted turn ID makes a duplicate event a no-op. Do not clear `stopBlockCount`, `stopMetricCursor`, or work-phase progress. For a fresh cwd without a session file, honor #255: no reset write; first verified mutation creates default state, and a later genuine prompt stamps the next turn. +In `handleUserPromptSubmit` at `:629-653`, read state before branch-specific writes. After the existing memory-write marker step (which may update state), re-read state, then reset only when `turn !== ""`, the state file already exists, and `state.stopBlockTurnId !== turn`. Persist `{ ...state, stopBlockTotal: 0, stopBlockTurnId: turn }`, then use this fresh snapshot in the rest of the handler. Do the reset before the `injectedTurns` dedupe, but the persisted turn ID makes a duplicate event a no-op. Do not clear `stopBlockCount`, `stopMetricCursor`, or work-phase progress. SessionStart normally creates the default state file. A direct UserPromptSubmit without prior SessionStart may still see no file; skip the reset in that case. No new state-creation API is added by #255. ```ts let state = readState(payload.cwd, payload.session_id); -if (turn && sessionStateFileExists(payload.cwd, payload.session_id) && state.stopBlockTurnId !== turn) { +if (turn && existsSync(statePath(payload.cwd, payload.session_id)) && state.stopBlockTurnId !== turn) { state = { ...state, stopBlockTotal: 0, stopBlockTurnId: turn }; writeState(payload.cwd, state); } if (turn && state.injectedTurns.includes(turn)) return ""; ``` -Use `sessionStateFileExists` added in `011`; do not infer existence from `readState`, which returns a default for absent files (`state.ts:486-603`). Place the reset after `hook.ts:643-651` memory marker so a later spread cannot overwrite either field. `bumpStopCounter` at `hook.ts:1459-1479` continues to increment `stopBlockTotal` for each Stop and release when `nextTotal > MAX_STOP_BLOCKS_TOTAL`. Change its return to distinguish `"phase-cap"` and `"total-cap"`, so only the absolute-cap release emits a message. Every caller at `hook.ts:1791,1810` must handle either release code. +Import `existsSync` from `node:fs` and the newly exported `statePath` from `state.ts`; use the file-existence guard before `writeState` because `readState` returns a default for absent files (`state.ts:412-420,486-603`). This is a narrow reset guard, not a new strict inspection or conditional-publication API. Place the reset after `hook.ts:643-651` memory marker so a later spread cannot overwrite either field. `bumpStopCounter` at `hook.ts:1459-1479` continues to increment `stopBlockTotal` for each Stop and release when `nextTotal > MAX_STOP_BLOCKS_TOTAL`. Change its return to distinguish `"phase-cap"` and `"total-cap"`, so only the absolute-cap release emits a message. Every caller at `hook.ts:1791,1810` must handle either release code. ```diff -if (nextCount > MAX_STOP_BLOCKS || nextTotal > MAX_STOP_BLOCKS_TOTAL) { diff --git a/devlog/_plan/260927_issue_train/015_issue251_worker_gate.md b/devlog/_plan/260927_issue_train/015_issue251_worker_gate.md index 632b98de..30adf1df 100644 --- a/devlog/_plan/260927_issue_train/015_issue251_worker_gate.md +++ b/devlog/_plan/260927_issue_train/015_issue251_worker_gate.md @@ -1,6 +1,6 @@ # #251 — Gate legacy worker only during an armed PABCD build/check cycle -Registered `executor` always needs an evidence receipt. A built-in `worker` needs one only when its parent session is actively orchestrating phase B or C. An ordinary worker releases at its first SubagentStop, without attempt files or tombstones. +When PABCD policy is enabled (012), registered `executor` needs an evidence receipt at every SubagentStop, regardless of parent state-file existence. SessionStart normally creates state; direct hook calls without it keep the existing executor fail-safe behavior. A built-in `worker` needs one only when its parent session is actively orchestrating phase B or C. With PABCD disabled, the gate is silent for **both** types and creates no attempt file or tombstone. An ordinary worker outside an armed B/C cycle releases at its first SubagentStop. Current anchors: `plugins/codexclaw/components/pabcd-state/src/subagent-evidence.ts:471`, `plugins/codexclaw/components/pabcd-state/src/state.ts:486`, `plugins/codexclaw/components/pabcd-state/test/subagent-evidence.test.ts:57`, `plugins/codexclaw/skills/pabcd/references/delegation.md:12`, `plugins/codexclaw/components/subagent-config/src/spawn-wrapper.ts:27`. @@ -10,6 +10,8 @@ Current anchors: `plugins/codexclaw/components/pabcd-state/src/subagent-evidence `GATED_AGENT_TYPES` currently includes both roles (`:57-64`) and `runSubagentStopGate` enters receipt logic immediately (`:471-510`). Keep the matcher set so registered hooks still reach this function; add the worker predicate before `extractReceiptPath`, `readAttempts`, or any state/evidence write. +The dispatch-level `readPabcdEnabled` switch in 012 runs first. For a direct call to this function, also check policy at the top of `runSubagentStopGate` (or inject an already-computed policy flag) so disabled behavior is consistent. Add `if (!readPabcdEnabled(payload.cwd)) return "";` before the role check, receipt extraction, attempts, tombstones and state writes. Test direct calls and built-hook dispatch. Do not add a parent-state-existence release clause; keep the existing executor attempts, tombstones, and completion consequences. + ```ts if (!GATED_AGENT_TYPES.has(payload.agent_type)) return ""; if (payload.agent_type === "worker") { @@ -18,15 +20,17 @@ if (payload.agent_type === "worker") { } ``` -`readStateStrict` is the existing non-throwing strict reader (`state.ts:486-603`); `orchestrationActive` is reconstructed false at IDLE (`state.ts:529`). The chosen predicate is **both** active orchestration and phase B/C. Phase P/A reviewers and ordinary worker delegations stay outside this receipt gate; only actual build/check workers need the legacy fallback. `executor` bypasses the predicate and retains the existing receipt, retry, tombstone, and parent-completion consequences. An unreadable worker state releases because the parent cycle cannot be proved armed; executor still follows the current fail-safe/tombstone behavior. Keep all existing receipt root validation for gated paths. +`readStateStrict` is the existing non-throwing strict reader (`state.ts:486-603`); `orchestrationActive` is reconstructed false at IDLE (`state.ts:529`). The chosen worker predicate is **both** active orchestration and phase B/C. Phase P/A reviewers and ordinary worker delegations stay outside this receipt gate. When policy is enabled, `executor` bypasses the worker predicate and retains the existing receipt, retry, tombstone, and parent-completion consequences. When policy is disabled, both roles release without touching evidence state. An unreadable worker state releases because the parent cycle cannot be proved armed; executor still follows the enabled-policy fail-safe behavior, with no 011 state-existence exception. Keep receipt root validation for gated paths. ### MODIFY `plugins/codexclaw/components/pabcd-state/test/subagent-evidence.test.ts` -At `:57-64`, replace the unarmed default-worker expectation with `worker outside armed PABCD releases without attempts or tombstone`: assert output `""`, `readAttempts(...) === 0`, no `.codexclaw/evidence-attempts` path, and `readState(...).unverifiedSubagents` empty. It fails before the fix because first Stop blocks. For the existing receipt/tombstone tests that use the test helper's default worker, either seed `{ phase:"B", orchestrationActive:true }` in each fixture or change only the helper default to `executor`; preserve explicit worker coverage. Add `worker in armed B and C blocks without receipt`, `worker at P/A/IDLE or orchestrationActive false releases`, `executor without active cycle still blocks`, and `worker with unreadable state releases without write`. Assert no attempt/tombstone writes on every release path. The `GATED_AGENT_TYPES` set assertion can remain (`test/:57` and later role checks): it describes hook routing, not unconditional gate application. +At `:57-64`, replace the unarmed default-worker expectation with `worker outside armed PABCD releases without attempts or tombstone`: assert output `""`, `readAttempts(...) === 0`, no `.codexclaw/evidence-attempts` path, and `readState(...).unverifiedSubagents` empty. It fails before the fix because first Stop blocks. For the existing receipt/tombstone tests that use the test helper's default worker, either seed `{ phase:"B", orchestrationActive:true }` in each fixture or change only the helper default to `executor`; preserve explicit worker coverage. Add `worker in armed B and C blocks without receipt`, `worker at P/A/IDLE or orchestrationActive false releases`, `executor with or without state and no active cycle still blocks`, and `worker with unreadable state releases without write`. Assert no attempt/tombstone writes on every release path. The `GATED_AGENT_TYPES` set assertion can remain (`test/:57` and later role checks): it describes hook routing, not unconditional gate application. + +Add `disabled PABCD releases executor and worker with no attempts/tombstones` for direct `runSubagentStopGate` calls and built-hook dispatch, including an otherwise armed B/C state. Add `enabled PABCD overrides project false and gates executor` to prove 012's positive env override. For a direct executor SubagentStop on a fresh cwd, retain the current evidence-gate result; its first directory creation receives 011's `.gitignore`. A fresh unarmed worker releases without attempts or a new directory. ### MODIFY `plugins/codexclaw/skills/pabcd/references/delegation.md` -At `:8-19`, add one sentence after the registered-executor fallback line: “The registered executor is evidence-gated on every SubagentStop; the built-in worker fallback is evidence-gated only while the parent has an active PABCD B/C cycle. Outside that cycle the worker releases without a receipt.” +At `:8-19`, add one sentence after the registered-executor fallback line: “When PABCD policy is enabled, the registered executor is evidence-gated on every SubagentStop; the built-in worker fallback is evidence-gated only while the parent has an active PABCD B/C cycle. When PABCD policy is disabled, both gates are silent. Outside that cycle the worker releases without a receipt.” ### MODIFY `plugins/codexclaw/components/subagent-config/src/spawn-wrapper.ts` @@ -34,4 +38,4 @@ Update the comment at `:8-12` to mention this distinction. Do not change `ROLE_A ## Activation and bypass record -Exercise executor with no state, worker with no state, worker P/A/B/C/IDLE, B/C with `orchestrationActive=false`, corrupt state, valid/invalid receipt, repeated Stop, and tombstone terminal behavior. Tier: cooperative SubagentStop enforcement; executing surface: `runSubagentStopGate`; known bypass: a child labeled as an ungated type or a missing/untrusted hook; residual risk: a worker that performs writes outside an armed cycle is no longer receipt-gated. Wording: “legacy worker receipts are required in active B/C cycles,” not “every worker is verified.” Final enforcement layer: SubagentStop runtime gate, followed by parent completion gate for recorded tombstones. Out of scope: changing native agent registration or receipt file format. +Exercise executor with no state (still gated), worker with no state (released), worker P/A/B/C/IDLE, B/C with `orchestrationActive=false`, both roles with PABCD off (including armed B/C), corrupt state, valid/invalid receipt, repeated Stop, and tombstone terminal behavior. **Tier:** SubagentStop `decision:block` is runtime continuation control analogous to E2, but structure/40 defines E2 for the `Stop` hook specifically; E8 tests cover this gate. **Executing surface:** `runSubagentStopGate` and hook dispatch. **Known bypass:** child labeled as an ungated type or a missing/untrusted hook. **Residual risk:** worker writes outside armed B/C are not receipt-gated. **Wording downgrade:** “registered executor receipts under enabled PABCD; legacy worker receipts in active B/C,” not “every worker is verified.” **Final enforcement layer:** SubagentStop runtime gate under enabled policy, then parent completion gate for recorded tombstones, with E8 regression tests. Out of scope: changing native agent registration or receipt file format. diff --git a/devlog/_plan/260927_issue_train/016_issue250_trigger_narrowing.md b/devlog/_plan/260927_issue_train/016_issue250_trigger_narrowing.md index a142aed9..ae146016 100644 --- a/devlog/_plan/260927_issue_train/016_issue250_trigger_narrowing.md +++ b/devlog/_plan/260927_issue_train/016_issue250_trigger_narrowing.md @@ -8,7 +8,9 @@ Current anchors: `plugins/codexclaw/components/pabcd-state/src/hook.ts:238`, `pl ### MODIFY `plugins/codexclaw/components/pabcd-state/src/hook.ts` -`detectTrigger` at `:238-250` currently matches bare `interview`/`인터뷰`, generic `plan this`, `build this`, Korean `구현해`/`검증해`, and `audit this`. Replace with a line-oriented explicit marker grammar. An accepted line must name `cxc-pabcd`, `codexclaw:cxc-pabcd`, a `[$cxc-pabcd](skill://...)` mention, `PABCD로`/`PABCD phase`, or a literal line-start `orchestrate [ipabc]`; for phase selection, it must additionally carry an unambiguous phase token (`interview/I`, `plan/P`, `audit/A`, `build/B`, `check/C`) on that line. A line-start `orchestrate` token remains handled by the existing parser first (`hook.ts:655-664`); the detector may recognize it for pure-function compatibility but must not broaden command parsing. Do not scan the whole prompt with an unanchored regex: quoted issue bodies often contain exact tokens. Add this complete helper before both detectors and replace `detectTrigger` with: +`detectTrigger` at `:238-250` currently matches bare `interview`/`인터뷰`, generic `plan this`, `build this`, Korean `구현해`/`검증해`, and `audit this`. Replace with a line-oriented explicit marker grammar. An accepted line must name `cxc-pabcd`, `codexclaw:cxc-pabcd`, a `[$cxc-pabcd](skill://...)` mention, `PABCD로`/`PABCD phase`, or a literal line-start `orchestrate [ipabc]`; for phase selection, it must additionally carry an unambiguous phase token (`interview/I`, `plan/P`, `audit/A`, `build/B`, `check/C`) on that line. A line-start `orchestrate` token remains handled by the existing parser first (`hook.ts:655-664`); the detector may recognize it for pure-function compatibility but must not broaden command parsing. Do not scan the whole prompt with an unanchored regex: quoted issue bodies often contain exact tokens. Strip inline spans enclosed by `"..."`, `“...”` or `'...'` before matching either detector. A quoted request is data, even when it appears after an imperative such as “Summarize.” Backtick spans are stripped too. Preserve a span whose whole content is a CodexClaw command token only for a direct execution request on that same line, under the imperative and explanatory-frame rules below. Tests include the two positive execution requests and the README explanation negative. Add this helper before both detectors and replace `detectTrigger` with the code below. + +The preserved backtick token is an execution request only when `run`, `use`, `start`, `invoke`, `실행`, `돌려`, `써서`, or `으로` directly addresses it on the same line. Explanatory framing (`explain`, `how to`, `what is`, `describe`, `README`, `docs`, `설명`, `어떻게`, `뭐야`) rejects the line. Add the negative ``Explain how to run `cxc-loop` from the README`` beside the positives ``Run `cxc-loop` for this task`` and ``Use `cxc-pabcd` to start Plan phase``. The code below applies this rule. ```ts function requestLines(prompt: string): string[] { @@ -18,7 +20,21 @@ function requestLines(prompt: string): string[] { const line = raw.trim(); if (/^```/.test(line)) { fenced = !fenced; continue; } if (fenced || !line || /^(?:>|[-*] |\d+[.)] )/.test(line)) continue; - result.push(line); + // Remove complete inline quotations before looking for marker or action. + // Treat apostrophes inside words as prose, not opening quotes. + const explanatory = /\b(?:explain|how to|what is|describe|readme|docs)\b|(?:설명|어떻게|뭐야)/i.test(line); + const unquoted = line + .replace(/`([^`]*)`/g, (match, inner: string, offset: number) => { + if (explanatory || !/^(?:\$?(?:codexclaw:)?cxc-(?:loop|pabcd)|orchestrate\s+[ipabc])$/i.test(inner.trim())) return " "; + const before = line.slice(0, offset); + const after = line.slice(offset + match.length); + const addressed = /(?:\b(?:run|use|start|invoke)\b|(?:실행|돌려|써서|으로))\s*$/i.test(before) + || /^\s*(?:으로|써서|실행|돌려)/.test(after); + return addressed ? inner : " "; + }) + .replace(/"(?:\\.|[^"\\])*"|“[^”]*”|(?/.codexclaw` itself; the separate existing SessionStart bootstrapping hook still creates state and 011's `.gitignore` when PABCD policy is enabled. The module itself only reads files and returns JSON and does not call `handleSessionStart`; the shared CLI path still records a hook observation under `CODEX_HOME` (plugins/codexclaw/scripts/hook-observation.mjs:17,70), never under the cwd. Both verbs stay active when `CODEXCLAW_PABCD=off` or `pabcd.enabled=false`, because they are not PABCD policy; 012's switch must not list them. - Success: with the explicit global opt-in, an agent-created root thread whose hook says `permission_mode: "default"` and whose user Codex config shows explicit top-level evidence `approval_policy = "never"` and `sandbox_mode = "danger-full-access"` (no profile) emits the exact allow object for Bash, write_stdin, apply_patch, request_permissions, and tool names that follow the `mcp____` naming convention. This is config evidence of user intent, not proof of the thread's effective permission; an exact guarantee needs Codex to expose the resolved policy and sandbox in PermissionRequest input. Every missing, mismatched, corrupt or unknown input emits zero stdout bytes and exits 0. The advisory is independent of opt-in. - Runtime proof boundary: hook input has `session_id`, `transcript_path`, `permission_mode`, `tool_name`, and optional `agent_id`/`agent_type` at `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/schema.rs:298-318`; SessionStart input has the first three at `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/schema.rs:496-510`. `SessionMeta` stores `id` and `thread_source` at `/tmp/cxc-perm/codex-src/codex-rs/protocol/src/protocol.rs:3128-3154`, and `ThreadSource::Feature` serializes its feature string at `/tmp/cxc-perm/codex-src/codex-rs/protocol/src/protocol.rs:2841-2857`. The specific `agent_created_thread` value is the observed rollout fixture from this issue train, not a universal enum variant. @@ -187,7 +187,7 @@ Current `readStdin` and overflow handling are at lines 68-109 and 333-340; `reco recordHookInvocation(raw, "pabcd-state", event, import.meta.url); ``` -No other event flow changes. In particular, the existing generic `session-start` handler remains side-effect-only at `cli.ts:406-410`; the new advisory does not mutate `.codexclaw` session state. +No other event flow changes. The existing generic `session-start` handler remains side-effect-only at `cli.ts:406-410` and continues to create session state when PABCD policy is enabled; the new advisory does not create `.codexclaw` itself. ### 3. NEW `plugins/codexclaw/hooks/permission-request-allowing-agent-thread.json` @@ -270,6 +270,8 @@ The existing real-hook golden tests at lines 53-76 and `EVENT_LABELS.PermissionR ## Runtime decision and bypass record +**Tier:** the synchronous PermissionRequest allow is a host permission decision **outside** structure/40's E1-E8 ladder (E1 means PreToolUse deny, not PermissionRequest allow); the SessionStart advisory is E4 context, and source tests/trust checks are E8. An allow does not override a separate denial. **Executing surface:** `handleAgentThreadPermissionRequest` and `handleAgentThreadSessionStartAdvisory` through their trusted hook manifests and `pabcd-state` CLI dispatch. **Known bypass:** an untrusted or absent hook has no effect; other hooks can deny; a direct caller can bypass the module predicates. **Residual risk:** runtime permission mode and sandbox/network constraints may differ from the top-level `config.toml` heuristic, and the `^mcp__` name filter may misclassify tools. **Wording downgrade:** “may suppress this approval prompt when all predicates hold,” not “agent threads have full access.” **Final enforcement layer:** Codex's PermissionRequest decision aggregator (`/tmp/cxc-perm/codex-src/codex-rs/hooks/src/events/permission_request.rs:149-169,301-327`); the advisory has no enforcement layer beyond context delivery. E8 tests check hook trust, allow bytes, and fail-open cases. + The PermissionRequest surface can suppress a user approval prompt only after all predicates pass. An empty stdout and exit 0 is no decision, so Codex keeps its normal prompt path; this is supported by `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/events/permission_request.rs:205-211`. The exact allow is parsed at `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/engine/output_parser.rs:184-204`. Never emit a denial, exit 2, `updatedInput`, `updatedPermissions`, or `interrupt`: the latter fields are unsupported at `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/schema.rs:199-217`, and exit 2 can deny at `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/events/permission_request.rs:249-263`. Another PermissionRequest hook's deny wins over this allow at `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/events/permission_request.rs:149-169,301-327`. Residual risks to record in the implementation PR: top-level `config.toml` can differ from runtime overrides; `permission_mode:"default"` does not prove the active sandbox; an allow result does not widen filesystem or network permissions; the `^mcp__` name check is a naming heuristic; another hook may deny; a new/modified hook declaration changes its trust hash and needs re-approval. The advisory therefore says only that approval mode may have degraded. This workaround does not assert upstream #33282, #40793, or #41167 is fixed. diff --git a/devlog/_plan/260927_issue_train/030_wp4_goalplan_decisions.md b/devlog/_plan/260927_issue_train/030_wp4_goalplan_decisions.md index 48b16bea..5e466bc8 100644 --- a/devlog/_plan/260927_issue_train/030_wp4_goalplan_decisions.md +++ b/devlog/_plan/260927_issue_train/030_wp4_goalplan_decisions.md @@ -282,6 +282,10 @@ Reuse `phase`, `plan`, and `roundTrip` at `:31-40` and `:155-161`. Add these nam These tests must include the negative enforcement paths because a filtered `ready` result alone cannot prove that cursor or recovery cannot advance a linked phase (`goalplan.ts:1795-1877`, `:1933-1975`, `:2033-2050`). Run focused tests with `node plugins/codexclaw/scripts/test.mjs "plugins/codexclaw/components/pabcd-state/test/goalplan-public-surface.test.ts" "plugins/codexclaw/components/pabcd-state/test/work-phase-states.test.ts"`, then `npm run build` (the root `package.json:21-24` defines build/test but no standalone typecheck). Run `git diff --check` after implementation. These are future builder checks; this docs-only phase did not run code tests. +## Enforcement and bypass record + +**Tier:** E8 validation for plan quality and integrity; CLI write preconditions and scheduler predicates are local runtime checks outside the E1-E8 hook ladder. Any hook guidance about a waiting phase is E4 only. **Executing surface:** `goalplan.ts` reviver, integrity/ready/cursor/close/recovery helpers, `goalplan-cli.ts` locked lifecycle mutation, and the existing bound-goal completion gate. **Known bypass:** a caller can use raw library writes or edit the plan file directly; an `ask` record can be omitted after a host question; unrelated host actions are not paused. **Residual risk:** same-user plan tampering and host answers that are never recorded leave the model and plan out of sync; a manual `done` edit must be caught by integrity/E8 before completion. **Wording downgrade:** “linked work phases wait while a recorded decision is open,” not “the host is paused” or “questions are automatically captured.” **Final enforcement layer:** common readiness predicates in `goalplan.ts:983-1004,1795-1877,1933-1975,2033-2050` plus definition integrity and E8 validation at `:1311-1494`; no hook alone can enforce the wait. The negative cursor/recovery tests above are required proof of those layers. + ## Activation scenarios | Condition | Trigger and expected path | From c94bb04dc68e125266d1102975e2d875335188e1 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 00:11:32 +0900 Subject: [PATCH 04/90] docs(plan): fold audit round 4 (gitignore atomicity, parent ignore, trigger framing) --- .../260927_issue_train/011_issue255_codexclaw_gitignore.md | 7 +++++++ .../260927_issue_train/016_issue250_trigger_narrowing.md | 5 +++++ 2 files changed, 12 insertions(+) diff --git a/devlog/_plan/260927_issue_train/011_issue255_codexclaw_gitignore.md b/devlog/_plan/260927_issue_train/011_issue255_codexclaw_gitignore.md index 0aefd962..a345ebf3 100644 --- a/devlog/_plan/260927_issue_train/011_issue255_codexclaw_gitignore.md +++ b/devlog/_plan/260927_issue_train/011_issue255_codexclaw_gitignore.md @@ -49,3 +49,10 @@ The other `mkdirSync` results do not first-create `/.codexclaw`: `plan-cli. - Add focused first-writer tests for representative independent components (at least `cxc-ops` PostCompact recovery, bg-wake `ensureDir`, and a non-state PABCD writer); assert the same ignore file. No source/test edit occurs in this docs-only phase. **Verification for the implementation branch:** focused `node --test` on touched component test files, `npm run build`, `npm test`, `npm run gate`, and the inventory check named in 010. Re-run `rg -n 'mkdirSync' plugins/codexclaw/components/*/src` and account for every project-local writer after the patch. The fix is partial because the directory and session file still appear at SessionStart; no promise of a read-only session source is made. + + +## Audit round 4 amendments (supersede conflicting text above) + +1. **Parent ignore rules.** A nested `.gitignore` cannot re-include files under a directory that an ancestor ignores. The `!rules/*.md` re-include therefore works only when no ancestor ignores `.codexclaw/`. When a project root ignores `.codexclaw/` (this repository does, `.gitignore:4`), committing rules needs a root-level exception in that project; codexclaw never edits a project's root `.gitignore`. Tests: in a temp git repo with no parent rule, `git check-ignore` ignores `sessions/x.json` and `ledger.jsonl` and does not ignore `rules/a.md`; with a root `.codexclaw/` rule, both are ignored and the helper still writes its own file unchanged. +2. **Atomic first creation.** `ensureCodexclawDir(cwd)` never exposes a `.codexclaw` without its `.gitignore`. When `/.codexclaw` is absent it creates a staging directory `/.codexclaw.staging--`, writes `.gitignore` inside it, then `renameSync(staging, /.codexclaw)`. If the rename fails because the target now exists (EEXIST, ENOTEMPTY, EPERM on Windows after a concurrent winner), it removes only its own staging directory and continues with the existing folder. Any other failure removes the staging directory and rethrows to the caller's existing error path. An existing `.codexclaw` is never modified by the helper, so a user-created folder stays untouched. Tests: `failure before rename leaves neither .codexclaw nor staging` (inject a throwing writer), `concurrent winner keeps its .gitignore` (pre-create the target between staging and rename via an injected rename), `existing folder without .gitignore is left alone`. +3. **Module placement and build order.** Components do not import each other (no `../../` imports exist) and `build.mjs:75-90` compiles each component on its own. The helper therefore lives in one small self-contained file, `src/codexclaw-dir.ts`, copied byte-identically into every component that can be a first creator (pabcd-state, cxc-ops, bg-wake, subagent-config, messenger-bridge, per the inventory above). A new test `plugins/codexclaw/test/codexclaw-dir-copies.test.mjs` asserts the copies are byte-identical, so drift fails CI. Order of verification: `npm run build` first, then the focused tests, then the full suite, inventory with the measured total, and gate. diff --git a/devlog/_plan/260927_issue_train/016_issue250_trigger_narrowing.md b/devlog/_plan/260927_issue_train/016_issue250_trigger_narrowing.md index ae146016..933c3f1a 100644 --- a/devlog/_plan/260927_issue_train/016_issue250_trigger_narrowing.md +++ b/devlog/_plan/260927_issue_train/016_issue250_trigger_narrowing.md @@ -98,3 +98,8 @@ These prompts and the old false positives are recorded in issue #250. Add `issue ## Activation and bypass record Exercise each of nine issue prompts, explicit English/Korean markers, skill link, duplicate turn, every inline quote form, quoted line, fenced code, mixed incidental and explicit lines, and explicit orchestrate grammar. **Tier:** E4 injected directive with E8 detector tests, not an E1 denial. **Executing surface:** `handleUserPromptSubmit`. **Known bypass:** other installed skills or direct human/CLI orchestration can still enter PABCD. **Residual risk:** an unusual natural-language request lacking a marker no longer gets a hint. **Wording downgrade:** “automatic hints require an explicit CodexClaw request,” with no claim that all phase entry is disabled. **Final enforcement layer:** detector plus existing parser-first dispatch, backed by E8 regression tests. Out of scope: changing explicit CLI/chat command grammar, goal policy, or unrelated search-request detection. + + +## Audit round 4 amendment (supersedes the explanation filter above) + +Explanatory framing is detected from the request structure, not from words anywhere on the line. A preserved backtick command token counts as a request only when the line starts (after optional politeness such as "please" or "좀") with an execution verb addressed to it (`run`, `use`, `start`, `invoke`, `실행`, `돌려`, or the token followed by `로`/`으로`/`써서`). A line that starts with an explanatory lead (`explain`, `describe`, `how do`, `how to`, `what is`, `what does`, `why`, `설명`, `어떻게`, `뭐야`) is not a request. Words such as `docs` or `README` later in the line no longer matter. Tests: positives `Run \`cxc-loop\` to update docs` and `Use \`cxc-pabcd\` to plan the README`, plus the earlier two positives; negative `Explain how to run \`cxc-loop\` from the README`, plus the quoted-data negatives. From e4016f281048e8595a0166d0f77527620d75ebcd Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 00:12:26 +0900 Subject: [PATCH 05/90] docs(plan): record audit rounds and builder workspace rule --- devlog/_plan/260927_issue_train/000_plan.md | 2 ++ .../002_architect_consultation.md | 14 ++++++++++++++ 2 files changed, 16 insertions(+) diff --git a/devlog/_plan/260927_issue_train/000_plan.md b/devlog/_plan/260927_issue_train/000_plan.md index 84d60a0b..a8aa6273 100644 --- a/devlog/_plan/260927_issue_train/000_plan.md +++ b/devlog/_plan/260927_issue_train/000_plan.md @@ -41,6 +41,8 @@ OUT: Codex core/Desktop, host automation mutation handler (#213), SessionStart f Delivery: one ordinary PR per implementation work phase from a `codex/issue-train-wpN` branch into `dev`, each merged after its hosted CI passes and the next phase rebased onto the new `dev`. No native stacks. Main owns git, the FSM, integration and delivery; gpt-6-sol subagents draft docs, build within disjoint scopes or task-owned worktrees, and review. +Builder workspaces: for parallel builders inside a work phase, main creates one managed worktree per builder with `create_worktree` (ref = the phase branch) and dispatches a gpt-6-sol subagent that passes that worktree path as the shell `workdir` on every command and commits on its own `codex/issue-train-wpN-` branch there. Subagents inherit this session's full-access permission, which avoids the reduced-permission start that `create_thread` children can get (001, agent-created thread section). Main merges builder branches into the phase branch in the order the decade doc gives, then runs the phase gates in this checkout. + ## Issue acceptance mapping | Issue | Decision | Where it lands | diff --git a/devlog/_plan/260927_issue_train/002_architect_consultation.md b/devlog/_plan/260927_issue_train/002_architect_consultation.md index 4aa073ce..a0f4155f 100644 --- a/devlog/_plan/260927_issue_train/002_architect_consultation.md +++ b/devlog/_plan/260927_issue_train/002_architect_consultation.md @@ -41,3 +41,17 @@ No further module-ownership conflicts were found across 010-016, 020 and 030. Three audit rounds returned **FAIL**. The root cause was the proposed lazy SessionStart state creation: stateless sessions would change the executor evidence gate and the goal-complete gate (`plugins/codexclaw/components/pabcd-state/src/goal-gate.ts:215-236`), and the proposed `writeExistingState` guard did not settle atomic conditional publication against `writeState`'s temporary-file rename (`state.ts:607-617`). The earlier no-parent-state executor release was especially unsafe. Main changed #255 to a low-severity **partial fix** using the reporter's offered `.codexclaw/.gitignore` alternative. SessionStart and CLI identity behavior stay as shipped; `011_issue255_codexclaw_gitignore.md` now routes first-directory writers through one exclusive ignore-file helper. Lazy state creation, `verifiedCreateState`, `sessionStateFileExists`, `writeExistingState`, missing-state hook suppression, and no-parent-state evidence release were dropped from this train and deferred for a separate design. The revised 010/013/014/015/020 contracts follow that decision. + + +## Audit record + +Reviewer handle `01a0e340-8900-7930-99db-9101bd95d661` (gpt-6-sol, CXC-ROLE: reviewer) audited every round. + +| Round | Verdict | Blockers | Disposition | +|---|---|---|---| +| 1 | FAIL | 8 (4 High) | All folded into 011, 012, 015, 016, 020, 030 | +| 2 | FAIL | 4 (1 High) | Folded; narrowed native verification to missing-state creation | +| 3 | FAIL | 3 (1 High) | Root cause: lazy state creation reaches the executor evidence gate, the goal-complete gate and publication atomicity. Returned to P and re-planned #255 as a partial fix | +| 4 | GO-WITH-FIXES | 4 (Medium) | Folded as "Audit round 4" amendments in 011 (parent ignore limit, atomic staging rename, self-contained helper copies with drift test, build order) and 016 (structural explanatory framing) | + +A>B exited as near-pass on round 4 with every blocker folded. From e486f8335c32af1368322b07c8e8f10c9a77043a Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 00:16:42 +0900 Subject: [PATCH 06/90] docs(plan): fold wp2 architect consultation --- .../_plan/260927_issue_train/002_architect_consultation.md | 5 +++++ .../260927_issue_train/011_issue255_codexclaw_gitignore.md | 5 +++++ devlog/_plan/260927_issue_train/012_issue252_pabcd_switch.md | 5 +++++ devlog/_plan/260927_issue_train/014_issue254_turn_budget.md | 5 +++++ 4 files changed, 20 insertions(+) diff --git a/devlog/_plan/260927_issue_train/002_architect_consultation.md b/devlog/_plan/260927_issue_train/002_architect_consultation.md index a0f4155f..8e5d0839 100644 --- a/devlog/_plan/260927_issue_train/002_architect_consultation.md +++ b/devlog/_plan/260927_issue_train/002_architect_consultation.md @@ -55,3 +55,8 @@ Reviewer handle `01a0e340-8900-7930-99db-9101bd95d661` (gpt-6-sol, CXC-ROLE: rev | 4 | GO-WITH-FIXES | 4 (Medium) | Folded as "Audit round 4" amendments in 011 (parent ignore limit, atomic staging rename, self-contained helper copies with drift test, build order) and 016 (structural explanatory framing) | A>B exited as near-pass on round 4 with every blocker folded. + + +## wp2 consultation (handle 01a0e36e-38aa-79e0-b94c-de2b217e695f) + +Proposal W2-1..W2-6 with reflection MISALIGNED on three partial mappings; W2-3, W2-5, W2-6 aligned. Main dispositions: W2-1 accepted, 011 now publishes with non-recursive `mkdirSync` plus empty-folder recovery; W2-2 accepted, 012 keeps the memory-request marker under the switch; W2-4 accepted, 014 makes the cap notice one-shot per turn. Merge rule accepted: branches may be built in parallel, merges are serial in 010's order, with test reconciliation after each merge. diff --git a/devlog/_plan/260927_issue_train/011_issue255_codexclaw_gitignore.md b/devlog/_plan/260927_issue_train/011_issue255_codexclaw_gitignore.md index a345ebf3..92e5c8e9 100644 --- a/devlog/_plan/260927_issue_train/011_issue255_codexclaw_gitignore.md +++ b/devlog/_plan/260927_issue_train/011_issue255_codexclaw_gitignore.md @@ -56,3 +56,8 @@ The other `mkdirSync` results do not first-create `/.codexclaw`: `plan-cli. 1. **Parent ignore rules.** A nested `.gitignore` cannot re-include files under a directory that an ancestor ignores. The `!rules/*.md` re-include therefore works only when no ancestor ignores `.codexclaw/`. When a project root ignores `.codexclaw/` (this repository does, `.gitignore:4`), committing rules needs a root-level exception in that project; codexclaw never edits a project's root `.gitignore`. Tests: in a temp git repo with no parent rule, `git check-ignore` ignores `sessions/x.json` and `ledger.jsonl` and does not ignore `rules/a.md`; with a root `.codexclaw/` rule, both are ignored and the helper still writes its own file unchanged. 2. **Atomic first creation.** `ensureCodexclawDir(cwd)` never exposes a `.codexclaw` without its `.gitignore`. When `/.codexclaw` is absent it creates a staging directory `/.codexclaw.staging--`, writes `.gitignore` inside it, then `renameSync(staging, /.codexclaw)`. If the rename fails because the target now exists (EEXIST, ENOTEMPTY, EPERM on Windows after a concurrent winner), it removes only its own staging directory and continues with the existing folder. Any other failure removes the staging directory and rethrows to the caller's existing error path. An existing `.codexclaw` is never modified by the helper, so a user-created folder stays untouched. Tests: `failure before rename leaves neither .codexclaw nor staging` (inject a throwing writer), `concurrent winner keeps its .gitignore` (pre-create the target between staging and rename via an injected rename), `existing folder without .gitignore is left alone`. 3. **Module placement and build order.** Components do not import each other (no `../../` imports exist) and `build.mjs:75-90` compiles each component on its own. The helper therefore lives in one small self-contained file, `src/codexclaw-dir.ts`, copied byte-identically into every component that can be a first creator (pabcd-state, cxc-ops, bg-wake, subagent-config, messenger-bridge, per the inventory above). A new test `plugins/codexclaw/test/codexclaw-dir-copies.test.mjs` asserts the copies are byte-identical, so drift fails CI. Order of verification: `npm run build` first, then the focused tests, then the full suite, inventory with the measured total, and gate. + + +## wp2 architect amendment (W2-1) + +`renameSync(staging, target)` can replace an existing empty directory on POSIX, so a directory rename does not give no-replace semantics. Publication instead uses `mkdirSync(target)` without `recursive`, which fails with EEXIST when the target exists: when it succeeds, this caller created the folder and immediately writes `.gitignore` with flag `wx`; when it fails with EEXIST, the folder belongs to someone else and is left alone. If the `.gitignore` write fails after a successful `mkdirSync`, the helper removes the directory it just created only when it is still empty (`rmdirSync`, which refuses a non-empty directory) and rethrows. The window between the two calls is covered by the recovery rule: a later first-writer that finds `.codexclaw` present, empty and without `.gitignore` treats it as an unfinished creation and writes the file with `wx`; a non-empty folder without `.gitignore` is never touched. Tests: `empty pre-existing .codexclaw is completed with .gitignore`, `non-empty pre-existing .codexclaw is left alone`, `gitignore write failure removes the empty folder it created`, `EEXIST from mkdir leaves the other creator's folder unchanged`. This replaces the staging-directory rename in "Audit round 4 amendments" item 2. diff --git a/devlog/_plan/260927_issue_train/012_issue252_pabcd_switch.md b/devlog/_plan/260927_issue_train/012_issue252_pabcd_switch.md index 5512b2c2..efb6eb5e 100644 --- a/devlog/_plan/260927_issue_train/012_issue252_pabcd_switch.md +++ b/devlog/_plan/260927_issue_train/012_issue252_pabcd_switch.md @@ -93,3 +93,8 @@ Exercise every matrix row, root/subagent payloads, executor/worker SubagentStop ## Note from the roadmap reflection The wp3 hook verbs `permission-request` and `session-start-permission-advisory` (020) are not PABCD policy. The switch must not list them, and wp3's integration test asserts they still run with `CODEXCLAW_PABCD=off`. + + +## wp2 architect amendment (W2-2) + +The memory-write authorization marker written in `handleUserPromptSubmit` (`hook.ts:632-653`, consumed by `memory-write-gate.ts:30`) is not PABCD policy. With the switch off, `user-prompt-submit` still runs the memory-request marker step and only skips the PABCD branches that follow it (explicit chat orchestrate, loop arm, trigger hints, passive reinjection). Test: `PABCD off still authorizes a user-requested memory write`: with `CODEXCLAW_PABCD=off`, a prompt asking to remember something followed by the memory PreToolUse call is allowed, and a prompt without such a request still denies. diff --git a/devlog/_plan/260927_issue_train/014_issue254_turn_budget.md b/devlog/_plan/260927_issue_train/014_issue254_turn_budget.md index aa71cbe5..27174485 100644 --- a/devlog/_plan/260927_issue_train/014_issue254_turn_budget.md +++ b/devlog/_plan/260927_issue_train/014_issue254_turn_budget.md @@ -67,3 +67,8 @@ For `total-cap`, return `JSON.stringify({ systemMessage: "CodexClaw Stop continu ## Activation and bypass record Exercise absent turn ID, duplicate turn ID, new turn ID, old-schema state, missing state, same-turn continuation, per-phase cap, and absolute cap. Tier: PABCD Stop-hook limit. Executing surface: UserPromptSubmit bookkeeping plus Stop decision. Known bypass: direct state edits or a native implementation that routes internal response items as user input; residual risk: future Codex runtime routing drift. Wording: “24 blocks per observed genuine user turn on the verified runtime.” Final enforcement layer: Stop `bumpStopCounter`. Out of scope: changing the 24/3 constants, host model retry budgets, or native Codex code. + + +## wp2 architect amendment (W2-4) + +The total-cap message is one-shot per user turn. Persist `stopBlockCapNotified: string | null` (the turn id already notified) beside `stopBlockTurnId`, with the same default, strict reconstruction and serialization chain. On `total-cap`, emit the `systemMessage` only when `stopBlockCapNotified !== stopBlockTurnId`, then record it; later Stops in that turn return `""`. A new genuine turn resets `stopBlockTotal` and the notice becomes eligible again. Tests: Stop 25 returns the message, Stop 26 returns `""`, a new turn's 25th Stop returns the message again. From cc5bd48abc7a3e26ff4fc09d91362a00eb5fda2c Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 00:17:33 +0900 Subject: [PATCH 07/90] docs(plan): fold wp2 reflection edge cases --- .../260927_issue_train/011_issue255_codexclaw_gitignore.md | 3 +++ devlog/_plan/260927_issue_train/014_issue254_turn_budget.md | 3 +++ 2 files changed, 6 insertions(+) diff --git a/devlog/_plan/260927_issue_train/011_issue255_codexclaw_gitignore.md b/devlog/_plan/260927_issue_train/011_issue255_codexclaw_gitignore.md index 92e5c8e9..42c08c11 100644 --- a/devlog/_plan/260927_issue_train/011_issue255_codexclaw_gitignore.md +++ b/devlog/_plan/260927_issue_train/011_issue255_codexclaw_gitignore.md @@ -61,3 +61,6 @@ The other `mkdirSync` results do not first-create `/.codexclaw`: `plan-cli. ## wp2 architect amendment (W2-1) `renameSync(staging, target)` can replace an existing empty directory on POSIX, so a directory rename does not give no-replace semantics. Publication instead uses `mkdirSync(target)` without `recursive`, which fails with EEXIST when the target exists: when it succeeds, this caller created the folder and immediately writes `.gitignore` with flag `wx`; when it fails with EEXIST, the folder belongs to someone else and is left alone. If the `.gitignore` write fails after a successful `mkdirSync`, the helper removes the directory it just created only when it is still empty (`rmdirSync`, which refuses a non-empty directory) and rethrows. The window between the two calls is covered by the recovery rule: a later first-writer that finds `.codexclaw` present, empty and without `.gitignore` treats it as an unfinished creation and writes the file with `wx`; a non-empty folder without `.gitignore` is never touched. Tests: `empty pre-existing .codexclaw is completed with .gitignore`, `non-empty pre-existing .codexclaw is left alone`, `gitignore write failure removes the empty folder it created`, `EEXIST from mkdir leaves the other creator's folder unchanged`. This replaces the staging-directory rename in "Audit round 4 amendments" item 2. + + +EEXIST on the `.gitignore` write is success when the existing file's bytes equal the helper's content (another caller finished the same creation); only a different or unreadable file is left alone without error. Test: `interleaved second creator writes .gitignore first; first creator treats identical EEXIST as success`. diff --git a/devlog/_plan/260927_issue_train/014_issue254_turn_budget.md b/devlog/_plan/260927_issue_train/014_issue254_turn_budget.md index 27174485..0e9724ed 100644 --- a/devlog/_plan/260927_issue_train/014_issue254_turn_budget.md +++ b/devlog/_plan/260927_issue_train/014_issue254_turn_budget.md @@ -72,3 +72,6 @@ Exercise absent turn ID, duplicate turn ID, new turn ID, old-schema state, missi ## wp2 architect amendment (W2-4) The total-cap message is one-shot per user turn. Persist `stopBlockCapNotified: string | null` (the turn id already notified) beside `stopBlockTurnId`, with the same default, strict reconstruction and serialization chain. On `total-cap`, emit the `systemMessage` only when `stopBlockCapNotified !== stopBlockTurnId`, then record it; later Stops in that turn return `""`. A new genuine turn resets `stopBlockTotal` and the notice becomes eligible again. Tests: Stop 25 returns the message, Stop 26 returns `""`, a new turn's 25th Stop returns the message again. + + +The notice latch is a boolean `stopBlockCapNotified: boolean` (default false), reset to false together with `stopBlockTotal` when a new genuine turn arrives. It works the same with or without a `turn_id`: Stop 25 emits and sets it, Stop 26 sees it set. This replaces the turn-id-valued latch above. Tests: Stops 25 and 26 with a turn id, Stops 25 and 26 with no turn id, and a new turn re-arming the notice. From 4113d92641c6a03dbb5c5daafa70958246d38664 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 00:20:22 +0900 Subject: [PATCH 08/90] docs(plan): consolidate wp2 amendments (gitignore publication, UPS switch path, cap notice chain) --- .../011_issue255_codexclaw_gitignore.md | 10 +++++++--- .../012_issue252_pabcd_switch.md | 5 ++--- .../260927_issue_train/014_issue254_turn_budget.md | 14 ++++++++++---- 3 files changed, 19 insertions(+), 10 deletions(-) diff --git a/devlog/_plan/260927_issue_train/011_issue255_codexclaw_gitignore.md b/devlog/_plan/260927_issue_train/011_issue255_codexclaw_gitignore.md index 42c08c11..65050478 100644 --- a/devlog/_plan/260927_issue_train/011_issue255_codexclaw_gitignore.md +++ b/devlog/_plan/260927_issue_train/011_issue255_codexclaw_gitignore.md @@ -57,10 +57,14 @@ The other `mkdirSync` results do not first-create `/.codexclaw`: `plan-cli. 2. **Atomic first creation.** `ensureCodexclawDir(cwd)` never exposes a `.codexclaw` without its `.gitignore`. When `/.codexclaw` is absent it creates a staging directory `/.codexclaw.staging--`, writes `.gitignore` inside it, then `renameSync(staging, /.codexclaw)`. If the rename fails because the target now exists (EEXIST, ENOTEMPTY, EPERM on Windows after a concurrent winner), it removes only its own staging directory and continues with the existing folder. Any other failure removes the staging directory and rethrows to the caller's existing error path. An existing `.codexclaw` is never modified by the helper, so a user-created folder stays untouched. Tests: `failure before rename leaves neither .codexclaw nor staging` (inject a throwing writer), `concurrent winner keeps its .gitignore` (pre-create the target between staging and rename via an injected rename), `existing folder without .gitignore is left alone`. 3. **Module placement and build order.** Components do not import each other (no `../../` imports exist) and `build.mjs:75-90` compiles each component on its own. The helper therefore lives in one small self-contained file, `src/codexclaw-dir.ts`, copied byte-identically into every component that can be a first creator (pabcd-state, cxc-ops, bg-wake, subagent-config, messenger-bridge, per the inventory above). A new test `plugins/codexclaw/test/codexclaw-dir-copies.test.mjs` asserts the copies are byte-identical, so drift fails CI. Order of verification: `npm run build` first, then the focused tests, then the full suite, inventory with the measured total, and gate. +## wp2 final publication rule (supersedes "Audit round 4 amendments" item 2 and every earlier recovery text) -## wp2 architect amendment (W2-1) +`ensureCodexclawDir(cwd)`: -`renameSync(staging, target)` can replace an existing empty directory on POSIX, so a directory rename does not give no-replace semantics. Publication instead uses `mkdirSync(target)` without `recursive`, which fails with EEXIST when the target exists: when it succeeds, this caller created the folder and immediately writes `.gitignore` with flag `wx`; when it fails with EEXIST, the folder belongs to someone else and is left alone. If the `.gitignore` write fails after a successful `mkdirSync`, the helper removes the directory it just created only when it is still empty (`rmdirSync`, which refuses a non-empty directory) and rethrows. The window between the two calls is covered by the recovery rule: a later first-writer that finds `.codexclaw` present, empty and without `.gitignore` treats it as an unfinished creation and writes the file with `wx`; a non-empty folder without `.gitignore` is never touched. Tests: `empty pre-existing .codexclaw is completed with .gitignore`, `non-empty pre-existing .codexclaw is left alone`, `gitignore write failure removes the empty folder it created`, `EEXIST from mkdir leaves the other creator's folder unchanged`. This replaces the staging-directory rename in "Audit round 4 amendments" item 2. +1. `mkdirSync(join(cwd, ".codexclaw"))` without `recursive`. EEXIST means the folder already exists: return without touching it, whatever it contains, including an empty folder, a folder without `.gitignore`, or a symlink. No recovery or repair of existing folders happens anywhere, so no `lstat` policy is needed. +2. Only after a successful `mkdirSync`, write `.gitignore` with flag `wx`. EEXIST here is success when the existing file's bytes equal the helper's content (a concurrent creator of the same folder); any other EEXIST content is left alone without error. +3. If the write fails with another error, remove the folder this call created with `rmdirSync` (which refuses a non-empty folder), then rethrow. +Residual risk: a process that dies between steps 1 and 2 leaves an empty `.codexclaw` without `.gitignore`, and later writers will not repair it. This is accepted for a low-severity issue; a user can delete the empty folder. -EEXIST on the `.gitignore` write is success when the existing file's bytes equal the helper's content (another caller finished the same creation); only a different or unreadable file is left alone without error. Test: `interleaved second creator writes .gitignore first; first creator treats identical EEXIST as success`. +Tests: `fresh cwd gets .codexclaw/.gitignore with exact bytes`; `existing empty .codexclaw is left alone`; `existing .codexclaw symlink is left alone and its target gets nothing`; `existing .gitignore is never overwritten`; `identical EEXIST on the ignore write is success`; `ignore write failure removes the empty folder it created and rethrows`. diff --git a/devlog/_plan/260927_issue_train/012_issue252_pabcd_switch.md b/devlog/_plan/260927_issue_train/012_issue252_pabcd_switch.md index efb6eb5e..640b1322 100644 --- a/devlog/_plan/260927_issue_train/012_issue252_pabcd_switch.md +++ b/devlog/_plan/260927_issue_train/012_issue252_pabcd_switch.md @@ -94,7 +94,6 @@ Exercise every matrix row, root/subagent payloads, executor/worker SubagentStop The wp3 hook verbs `permission-request` and `session-start-permission-advisory` (020) are not PABCD policy. The switch must not list them, and wp3's integration test asserts they still run with `CODEXCLAW_PABCD=off`. +## wp2 final UserPromptSubmit rule (supersedes the W2-2 paragraph and the Set above) -## wp2 architect amendment (W2-2) - -The memory-write authorization marker written in `handleUserPromptSubmit` (`hook.ts:632-653`, consumed by `memory-write-gate.ts:30`) is not PABCD policy. With the switch off, `user-prompt-submit` still runs the memory-request marker step and only skips the PABCD branches that follow it (explicit chat orchestrate, loop arm, trigger hints, passive reinjection). Test: `PABCD off still authorizes a user-requested memory write`: with `CODEXCLAW_PABCD=off`, a prompt asking to remember something followed by the memory PreToolUse call is allowed, and a prompt without such a request still denies. +Remove `"user-prompt-submit"` from `PABCD_DISABLED_EVENTS`. In the `user-prompt-submit` dispatch branch, call `handleUserPromptSubmit(raw, { pabcdEnabled })`. In `hook.ts`, give `handleUserPromptSubmit` an optional second parameter `options: { pabcdEnabled?: boolean } = {}` and, immediately after the memory-request marker step (`hook.ts:643-653`), add `if (options.pabcdEnabled === false) return "";`. Everything after that line (explicit chat orchestrate, turn-budget reset, loop arm, trigger hints, search hint, passive reinjection) is PABCD policy and is skipped. Tests: `PABCD off still authorizes a user-requested memory write` (prompt asking to remember, then the memory PreToolUse call is allowed), `PABCD off ordinary prompt leaves memory writes denied`, `PABCD off prompt emits no PABCD context`. diff --git a/devlog/_plan/260927_issue_train/014_issue254_turn_budget.md b/devlog/_plan/260927_issue_train/014_issue254_turn_budget.md index 0e9724ed..4279b926 100644 --- a/devlog/_plan/260927_issue_train/014_issue254_turn_budget.md +++ b/devlog/_plan/260927_issue_train/014_issue254_turn_budget.md @@ -68,10 +68,16 @@ For `total-cap`, return `JSON.stringify({ systemMessage: "CodexClaw Stop continu Exercise absent turn ID, duplicate turn ID, new turn ID, old-schema state, missing state, same-turn continuation, per-phase cap, and absolute cap. Tier: PABCD Stop-hook limit. Executing surface: UserPromptSubmit bookkeeping plus Stop decision. Known bypass: direct state edits or a native implementation that routes internal response items as user input; residual risk: future Codex runtime routing drift. Wording: “24 blocks per observed genuine user turn on the verified runtime.” Final enforcement layer: Stop `bumpStopCounter`. Out of scope: changing the 24/3 constants, host model retry budgets, or native Codex code. +## wp2 final cap-notice rule (supersedes both W2-4 paragraphs that were here) -## wp2 architect amendment (W2-4) +Field chain for `stopBlockCapNotified: boolean`: -The total-cap message is one-shot per user turn. Persist `stopBlockCapNotified: string | null` (the turn id already notified) beside `stopBlockTurnId`, with the same default, strict reconstruction and serialization chain. On `total-cap`, emit the `systemMessage` only when `stopBlockCapNotified !== stopBlockTurnId`, then record it; later Stops in that turn return `""`. A new genuine turn resets `stopBlockTotal` and the notice becomes eligible again. Tests: Stop 25 returns the message, Stop 26 returns `""`, a new turn's 25th Stop returns the message again. +- Type: add to `State` beside `stopBlockTotal` and `stopBlockTurnId` (`state.ts:135-137`). +- Default: `false` in `defaultState` (`state.ts:299-301`). +- Reconstruction: `parsed.stopBlockCapNotified === true` in the strict reader beside `state.ts:551-554`; anything else is `false`. Old files read as `false`. +- Serialization: `writeState` writes the whole object (`state.ts:607-615`); nothing extra. +- Reset: the new-turn reset in `handleUserPromptSubmit` writes `{ stopBlockTotal: 0, stopBlockTurnId: turn, stopBlockCapNotified: false }`. +- Cap write: `bumpStopCounter` returns `"block" | "phase-cap" | "total-cap" | "total-cap-silent"`. On a total-cap release it writes `stopBlockCapNotified: true` in the same `writeState` call and returns `"total-cap"` when the flag was false before, `"total-cap-silent"` when it was already true. +- Caller output: both callers (`hook.ts:1791` IDLE path and `hook.ts:1810` in-flight path) map `"total-cap"` to `JSON.stringify({ systemMessage: "CodexClaw Stop continuation cap (24) reached for this user turn; releasing." }) + "\n"` and every other release code to `""`. - -The notice latch is a boolean `stopBlockCapNotified: boolean` (default false), reset to false together with `stopBlockTotal` when a new genuine turn arrives. It works the same with or without a `turn_id`: Stop 25 emits and sets it, Stop 26 sees it set. This replaces the turn-id-valued latch above. Tests: Stops 25 and 26 with a turn id, Stops 25 and 26 with no turn id, and a new turn re-arming the notice. +Tests in `hook-continuation.test.ts`: `cap notice once per turn (turn id)` (Stop 25 returns the message, Stop 26 returns ""), `cap notice once per turn (no turn id)` (same with no `turn_id` in any payload), `new turn re-arms the cap notice`; in `state.test.ts`: `stopBlockCapNotified defaults to false and round-trips`. From c048886de45112c6577c1e71f648303b5920d2b5 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 00:22:38 +0900 Subject: [PATCH 09/90] docs(plan): fix UPS handler signature and cap return contract --- .../011_issue255_codexclaw_gitignore.md | 2 +- .../012_issue252_pabcd_switch.md | 2 +- .../014_issue254_turn_budget.md | 17 ++++++++++++----- 3 files changed, 14 insertions(+), 7 deletions(-) diff --git a/devlog/_plan/260927_issue_train/011_issue255_codexclaw_gitignore.md b/devlog/_plan/260927_issue_train/011_issue255_codexclaw_gitignore.md index 65050478..5775618d 100644 --- a/devlog/_plan/260927_issue_train/011_issue255_codexclaw_gitignore.md +++ b/devlog/_plan/260927_issue_train/011_issue255_codexclaw_gitignore.md @@ -67,4 +67,4 @@ The other `mkdirSync` results do not first-create `/.codexclaw`: `plan-cli. Residual risk: a process that dies between steps 1 and 2 leaves an empty `.codexclaw` without `.gitignore`, and later writers will not repair it. This is accepted for a low-severity issue; a user can delete the empty folder. -Tests: `fresh cwd gets .codexclaw/.gitignore with exact bytes`; `existing empty .codexclaw is left alone`; `existing .codexclaw symlink is left alone and its target gets nothing`; `existing .gitignore is never overwritten`; `identical EEXIST on the ignore write is success`; `ignore write failure removes the empty folder it created and rethrows`. +Tests: `fresh cwd gets .codexclaw/.gitignore with exact bytes`; `existing empty .codexclaw is left alone`; `existing .codexclaw symlink is left alone and its target gets nothing` (calls the helper alone, not a full writer); `existing .gitignore is never overwritten`; `identical EEXIST on the ignore write is success`; `ignore write failure removes the empty folder it created and rethrows`. diff --git a/devlog/_plan/260927_issue_train/012_issue252_pabcd_switch.md b/devlog/_plan/260927_issue_train/012_issue252_pabcd_switch.md index 640b1322..c5a5ac71 100644 --- a/devlog/_plan/260927_issue_train/012_issue252_pabcd_switch.md +++ b/devlog/_plan/260927_issue_train/012_issue252_pabcd_switch.md @@ -96,4 +96,4 @@ The wp3 hook verbs `permission-request` and `session-start-permission-advisory` ## wp2 final UserPromptSubmit rule (supersedes the W2-2 paragraph and the Set above) -Remove `"user-prompt-submit"` from `PABCD_DISABLED_EVENTS`. In the `user-prompt-submit` dispatch branch, call `handleUserPromptSubmit(raw, { pabcdEnabled })`. In `hook.ts`, give `handleUserPromptSubmit` an optional second parameter `options: { pabcdEnabled?: boolean } = {}` and, immediately after the memory-request marker step (`hook.ts:643-653`), add `if (options.pabcdEnabled === false) return "";`. Everything after that line (explicit chat orchestrate, turn-budget reset, loop arm, trigger hints, search hint, passive reinjection) is PABCD policy and is skipped. Tests: `PABCD off still authorizes a user-requested memory write` (prompt asking to remember, then the memory PreToolUse call is allowed), `PABCD off ordinary prompt leaves memory writes denied`, `PABCD off prompt emits no PABCD context`. +Remove `"user-prompt-submit"` from `PABCD_DISABLED_EVENTS`. In the `user-prompt-submit` dispatch branch (`cli.ts:410-412`), keep the parsed payload and call `handleUserPromptSubmit(payload, process.platform, {}, { pabcdEnabled })`. In `hook.ts`, add a fourth optional parameter after the existing ones (`hook.ts:624-628`: `payload, platform, dcloseCommitHooks`): `options: { pabcdEnabled?: boolean } = {}`, so every existing caller and test keeps working unchanged, and, immediately after the memory-request marker step (`hook.ts:643-653`), add `if (options.pabcdEnabled === false) return "";`. Everything after that line (explicit chat orchestrate, turn-budget reset, loop arm, trigger hints, search hint, passive reinjection) is PABCD policy and is skipped. Tests: `PABCD off still authorizes a user-requested memory write` (prompt asking to remember, then the memory PreToolUse call is allowed), `PABCD off ordinary prompt leaves memory writes denied`, `PABCD off prompt emits no PABCD context`. diff --git a/devlog/_plan/260927_issue_train/014_issue254_turn_budget.md b/devlog/_plan/260927_issue_train/014_issue254_turn_budget.md index 4279b926..9ad34fb8 100644 --- a/devlog/_plan/260927_issue_train/014_issue254_turn_budget.md +++ b/devlog/_plan/260927_issue_train/014_issue254_turn_budget.md @@ -44,17 +44,24 @@ if (turn && existsSync(statePath(payload.cwd, payload.session_id)) && state.stop if (turn && state.injectedTurns.includes(turn)) return ""; ``` -Import `existsSync` from `node:fs` and the newly exported `statePath` from `state.ts`; use the file-existence guard before `writeState` because `readState` returns a default for absent files (`state.ts:412-420,486-603`). This is a narrow reset guard, not a new strict inspection or conditional-publication API. Place the reset after `hook.ts:643-651` memory marker so a later spread cannot overwrite either field. `bumpStopCounter` at `hook.ts:1459-1479` continues to increment `stopBlockTotal` for each Stop and release when `nextTotal > MAX_STOP_BLOCKS_TOTAL`. Change its return to distinguish `"phase-cap"` and `"total-cap"`, so only the absolute-cap release emits a message. Every caller at `hook.ts:1791,1810` must handle either release code. +Import `existsSync` from `node:fs` and the newly exported `statePath` from `state.ts`; use the file-existence guard before `writeState` because `readState` returns a default for absent files (`state.ts:412-420,486-603`). This is a narrow reset guard, not a new strict inspection or conditional-publication API. Place the reset after `hook.ts:643-651` memory marker so a later spread cannot overwrite either field. `bumpStopCounter` at `hook.ts:1459-1479` continues to increment `stopBlockTotal` for each Stop and release when `nextTotal > MAX_STOP_BLOCKS_TOTAL`. Its return contract and both callers are specified once, in "wp2 final cap-notice rule" at the end of this document (a numeric block count, or `phase-cap`, `total-cap`, `total-cap-silent`). ```diff -if (nextCount > MAX_STOP_BLOCKS || nextTotal > MAX_STOP_BLOCKS_TOTAL) { -+if (nextCount > MAX_STOP_BLOCKS || nextTotal > MAX_STOP_BLOCKS_TOTAL) { - writeState(cwd, { ...state, ...carry, stopBlockPhase: null, stopBlockWorkPhaseId: null, stopBlockCount: 0 }); +- writeState(cwd, { ...state, ...carry, stopBlockPhase: null, stopBlockWorkPhaseId: null, stopBlockCount: 0 }); - return "release"; -+ return nextTotal > MAX_STOP_BLOCKS_TOTAL ? "total-cap" : "phase-cap"; ++if (nextCount > MAX_STOP_BLOCKS || nextTotal > MAX_STOP_BLOCKS_TOTAL) { ++ const totalCap = nextTotal > MAX_STOP_BLOCKS_TOTAL; ++ const alreadyNotified = state.stopBlockCapNotified === true; ++ writeState(cwd, { ...state, ...carry, stopBlockPhase: null, stopBlockWorkPhaseId: null, stopBlockCount: 0, ++ stopBlockCapNotified: totalCap ? true : state.stopBlockCapNotified }); ++ if (!totalCap) return "phase-cap"; ++ return alreadyNotified ? "total-cap-silent" : "total-cap"; } ``` +The block path keeps returning the numeric `nextCount` as today (`hook.ts:1471-1478`); the function type becomes `number | "phase-cap" | "total-cap" | "total-cap-silent"` (`hook.ts:1459`), and callers treat any number as a block. + For `total-cap`, return `JSON.stringify({ systemMessage: "CodexClaw Stop continuation cap (24) reached for this user turn; releasing." })` plus newline. Do not include `decision:block`, `continue:false`, or `stopReason`. The universal Stop output accepts `systemMessage` (`/tmp/cxc-perm/codex-src/codex-rs/hooks/src/schema.rs:90-99,455-464`); the runtime records it as a Warning without blocking (`/tmp/cxc-perm/codex-src/codex-rs/hooks/src/events/stop.rs:277-293`). A phase-cap returns `""` as before. ### MODIFY tests @@ -77,7 +84,7 @@ Field chain for `stopBlockCapNotified: boolean`: - Reconstruction: `parsed.stopBlockCapNotified === true` in the strict reader beside `state.ts:551-554`; anything else is `false`. Old files read as `false`. - Serialization: `writeState` writes the whole object (`state.ts:607-615`); nothing extra. - Reset: the new-turn reset in `handleUserPromptSubmit` writes `{ stopBlockTotal: 0, stopBlockTurnId: turn, stopBlockCapNotified: false }`. -- Cap write: `bumpStopCounter` returns `"block" | "phase-cap" | "total-cap" | "total-cap-silent"`. On a total-cap release it writes `stopBlockCapNotified: true` in the same `writeState` call and returns `"total-cap"` when the flag was false before, `"total-cap-silent"` when it was already true. +- Cap write: `bumpStopCounter` returns `number | "phase-cap" | "total-cap" | "total-cap-silent"` (a number is a block, as today). On a total-cap release it writes `stopBlockCapNotified: true` in the same `writeState` call and returns `"total-cap"` when the flag was false before, `"total-cap-silent"` when it was already true. - Caller output: both callers (`hook.ts:1791` IDLE path and `hook.ts:1810` in-flight path) map `"total-cap"` to `JSON.stringify({ systemMessage: "CodexClaw Stop continuation cap (24) reached for this user turn; releasing." }) + "\n"` and every other release code to `""`. Tests in `hook-continuation.test.ts`: `cap notice once per turn (turn id)` (Stop 25 returns the message, Stop 26 returns ""), `cap notice once per turn (no turn id)` (same with no `turn_id` in any payload), `new turn re-arms the cap notice`; in `state.test.ts`: `stopBlockCapNotified defaults to false and round-trips`. From 9b09b0986e6eb899daeddf9fe4a821fb09f1820a Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 00:31:38 +0900 Subject: [PATCH 10/90] fix(pabcd-state): release unbound IDLE goals (#253) --- .../components/pabcd-state/src/hook.ts | 39 +++++------ .../test/hook-continuation.test.ts | 65 +++++++++++++------ 2 files changed, 63 insertions(+), 41 deletions(-) diff --git a/plugins/codexclaw/components/pabcd-state/src/hook.ts b/plugins/codexclaw/components/pabcd-state/src/hook.ts index 537df632..23b7a971 100644 --- a/plugins/codexclaw/components/pabcd-state/src/hook.ts +++ b/plugins/codexclaw/components/pabcd-state/src/hook.ts @@ -8,9 +8,9 @@ * * Stop: active under a native goal only. It returns a bounded * `{decision:"block",reason}` continuation envelope while a PABCD cycle is in flight — - * or, since 260709 (GOAL-IDLE-CONTINUE-01), while an ACTIVE goal is parked with no - * in-flight cycle (arming nudge). It releases on: no active goal, phase I, context - * pressure, or the same-phase stagnation cap (the single total-termination bound now + * or, since 260709 (GOAL-IDLE-CONTINUE-01), while an ACTIVE goal has a bound + * goalplan but no in-flight cycle (arming nudge). It releases on: no active goal, + * no bound plan, phase I, context pressure, or the same-phase stagnation cap (the single total-termination bound now * that the old unconditional `stop_hook_active` release is gone). * * Ground truth: @@ -1643,14 +1643,12 @@ export function readStopWorkContext(cwd: string, state: State): StopWorkContext } /** - * GOAL-IDLE-CONTINUE-01 (260709) — the Stop block for "goal ACTIVE but no PABCD cycle - * in flight". The old guard 2a released this state silently, so a session could park an - * active goal at IDLE forever (019f4407: goal created, FSM never entered, turn ended). + * GOAL-IDLE-CONTINUE-01 (260709) — the Stop block for a goal ACTIVE with a bound + * goalplan but no PABCD cycle in flight. An unbound goal releases at IDLE. * The reason names the two honest exits: arm the next work-phase (`orchestrate P`), or * close the goal for real (`update_goal complete` — gated by GOAL-COMPLETE-GATE-01 when - * a goalplan is bound — or `blocked` for external blockers). When a goalplan is bound, - * the remaining work is named; when it is bound but unregistered (empty), the block says - * to fill it; when none is bound, it points at `cxc loop init`. + * a goalplan is bound — or `blocked` for external blockers). The remaining work is + * named when present; an empty bound plan gets guidance to register work phases. */ export function buildGoalIdleBlock( cwd: string, @@ -1746,9 +1744,10 @@ function objectivePlateau(cwd: string, sessionId: string): PlateauCheck { /** * Stop handler — L6 active continuation with a bounded stagnation guard so the loop * ALWAYS terminates. Blocks (keeps the agent going) only when a PABCD cycle is genuinely - * in flight under an active goal, OR when an ACTIVE goal is parked with no in-flight - * cycle (GOAL-IDLE-CONTINUE-01: arming nudge). Releases via any of: no active goal, - * phase I (interview firewall), context pressure, or the MAX_STOP_BLOCKS cap. + * in flight under an active goal, OR when an ACTIVE goal with a bound plan is + * parked at IDLE (GOAL-IDLE-CONTINUE-01: arming nudge). Releases via any of: + * no active goal, no bound plan at IDLE, phase I (interview firewall), context + * pressure, or the MAX_STOP_BLOCKS cap. * * 260709 (lazygap loop-enforcement patch): * - guard 1 (`stop_hook_active` → unconditional release) is REMOVED. Under the old @@ -1757,13 +1756,10 @@ function objectivePlateau(cwd: string, sessionId: string): PlateauCheck { * phase progress, which is the "step-by-step cut" the loop doctrine forbids. * Termination stays total: the per-phase MAX_STOP_BLOCKS stagnation cap (reset on * every real transition) bounds every continuation chain that stops progressing. - * - GOAL-IDLE-CONTINUE-01: an ACTIVE goal with no in-flight cycle used to release - * silently (guard 2a), so "goal armed but PABCD never entered" (019f4407) ended - * turns freely. It now gets the same bounded block, naming the arming command - * (`cxc orchestrate P --session `), the goalplan's remaining work when one is - * bound, and the honest close-out path (update_goal complete gated by E8 / blocked). - * Side effect by design: the counter write creates the session state file, so the - * suggested orchestrate command passes the G2 unknown-session guard afterwards. + * - GOAL-IDLE-CONTINUE-01: an ACTIVE goal with a resolvable bound plan gets a + * bounded block naming the arming command (`cxc orchestrate P --session `), + * remaining work, and the honest close-out path (update_goal complete gated by + * E8 / blocked). An unbound or stale slug releases without a counter write. */ export function handleStop( payload: StopPayload, @@ -1782,10 +1778,11 @@ export function handleStop( const inFlight = state.orchestrationActive && state.phase !== "IDLE"; // guard 2a (amended by GOAL-IDLE-CONTINUE-01): with no cycle in flight a plain - // interactive session releases exactly as before; an ACTIVE goal instead gets a - // bounded arming block — "IDLE is not the end while work remains" (LOOP-CONTINUE-01). + // interactive session releases exactly as before; an ACTIVE goal with a bound + // plan gets a bounded arming block — "IDLE is not the end while work remains". if (!inFlight) { if (!goalActive) return ""; + if (!state.slug || !safeReadBoundGoalplan(payload.cwd, state.slug)) return ""; // bail: don't pile on during context-pressure/compaction recovery. if (isContextPressureTail(readTranscriptTail(payload.transcript_path))) return ""; if (bumpStopCounter(payload.cwd, state) === "release") return ""; diff --git a/plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts b/plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts index 341cff01..1f03a1c3 100644 --- a/plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts @@ -1,12 +1,13 @@ import { test } from "node:test"; import assert from "node:assert/strict"; -import { mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs"; +import { existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs"; import { createRequire } from "node:module"; import { tmpdir } from "node:os"; import { join } from "node:path"; import { buildStageHeader, handleUserPromptSubmit, + handleSessionStart, handleStop, MAX_STOP_BLOCKS, MAX_STOP_BLOCKS_TOTAL, @@ -501,29 +502,33 @@ test("L6: guard 2a — IDLE / inactive orchestration releases for a plain sessio } finally { rmSync(cwd, { recursive: true, force: true }); } }); -// ── 260709 GOAL-IDLE-CONTINUE-01: active goal + no in-flight cycle = bounded arming block ── +// ── GOAL-IDLE-CONTINUE-01: only a bound goalplan arms the IDLE block ── -test("GOAL-IDLE-CONTINUE-01: active goal at IDLE blocks with the arming command", () => { +test("GOAL-IDLE-CONTINUE-01: active goal without bound plan releases without state write", () => { const cwd = freshCwd(); try { withGoalsDb([{ thread_id: "gi1", status: "active" }], () => { - // no state file at all (019f4407 shape: goal created, FSM never entered) - const out = handleStop(stop(cwd, "gi1"), "linux"); - const parsed = JSON.parse(out.trim()); - assert.equal(parsed.decision, "block"); - assert.match(parsed.reason, /goal continuation/); - assert.match(parsed.reason, /GOAL-IDLE-CONTINUE-01/); - // `--attest` is a PREFIX of `--attest-file`, so the old assertion passed on - // win32 by accident. Pin the POSIX form explicitly; the win32 branch is - // asserted separately below. - assert.match(parsed.reason, /cxc orchestrate P --session gi1 --attest '\{/); - assert.match(parsed.reason, /update_goal/); - assert.match(parsed.reason, /LOOP-UNIT-CHAIN-01/, "IDLE block must teach heterogeneous work-phase chaining"); - assert.match(parsed.reason, /cxc loop init/, "unbound session must be pointed at loop init"); - // the counter write bootstraps the session file, keyed at IDLE - const st = readState(cwd, "gi1"); - assert.equal(st.stopBlockPhase, "IDLE"); - assert.equal(st.stopBlockCount, 1); + assert.equal(handleStop(stop(cwd, "gi1"), "linux"), ""); + assert.equal(existsSync(join(cwd, ".codexclaw")), false); + assert.equal(handleStop(stop(cwd, "gi1"), "linux"), ""); + assert.equal(existsSync(join(cwd, ".codexclaw")), false); + }); + } finally { rmSync(cwd, { recursive: true, force: true }); } +}); + +test("GOAL-IDLE-CONTINUE-01: SessionStart state and gitignore survive unbound IDLE Stop", () => { + const cwd = freshCwd(); + try { + withGoalsDb([{ thread_id: "gi-start", status: "active" }], () => { + assert.equal(handleSessionStart({ hook_event_name: "SessionStart", session_id: "gi-start", cwd }), ""); + const sessionPath = join(cwd, ".codexclaw", "sessions", "gi-start.json"); + const ignorePath = join(cwd, ".codexclaw", ".gitignore"); + writeFileSync(ignorePath, "pre-existing ignore\n"); + const stateBefore = readFileSync(sessionPath, "utf8"); + const ignoreBefore = readFileSync(ignorePath, "utf8"); + assert.equal(handleStop(stop(cwd, "gi-start")), ""); + assert.equal(readFileSync(sessionPath, "utf8"), stateBefore); + assert.equal(readFileSync(ignorePath, "utf8"), ignoreBefore); }); } finally { rmSync(cwd, { recursive: true, force: true }); } }); @@ -536,6 +541,9 @@ test("GOAL-IDLE-CONTINUE-01: the win32 block teaches the file flag, not inline a const cwd = freshCwd(); try { withGoalsDb([{ thread_id: "gi1", status: "active" }], () => { + const plan = buildGoalplan({ objective: "Win32 IDLE continuation" }); + writeGoalplan(cwd, plan); + writeState(cwd, { ...defaultState("gi1"), slug: plan.slug }); const parsed = JSON.parse(handleStop(stop(cwd, "gi1"), "win32").trim()); assert.equal(parsed.decision, "block"); assert.doesNotMatch(parsed.reason, /--attest '\{/); @@ -555,6 +563,9 @@ test("GOAL-IDLE-CONTINUE-01: bounded — releases after MAX_STOP_BLOCKS blocks a const cwd = freshCwd(); try { withGoalsDb([{ thread_id: "gi2", status: "active" }], () => { + const plan = buildGoalplan({ objective: "Bounded IDLE continuation" }); + writeGoalplan(cwd, plan); + writeState(cwd, { ...defaultState("gi2"), slug: plan.slug }); for (let i = 0; i < MAX_STOP_BLOCKS; i++) { assert.notEqual(handleStop(stop(cwd, "gi2")), "", `IDLE block ${i + 1} should block`); } @@ -599,6 +610,20 @@ test("GOAL-IDLE-CONTINUE-01: bound but EMPTY goalplan is told to register the pl } finally { rmSync(cwd, { recursive: true, force: true }); } }); +test("GOAL-IDLE-CONTINUE-01: stale slug releases without counter write", () => { + const cwd = freshCwd(); + try { + withGoalsDb([{ thread_id: "gi-stale", status: "active" }], () => { + writeState(cwd, { ...defaultState("gi-stale"), slug: "missing-plan", stopBlockTotal: 7 }); + const sessionPath = join(cwd, ".codexclaw", "sessions", "gi-stale.json"); + const before = readFileSync(sessionPath, "utf8"); + assert.equal(handleStop(stop(cwd, "gi-stale")), ""); + assert.equal(readState(cwd, "gi-stale").stopBlockTotal, 7); + assert.equal(readFileSync(sessionPath, "utf8"), before); + }); + } finally { rmSync(cwd, { recursive: true, force: true }); } +}); + test("L6: guard 2b — no active goal releases even mid-cycle (interactive pause)", () => { const cwd = freshCwd(); try { From 45eeaa0a39a63d36146c137792a5b3038477777e Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 00:33:53 +0900 Subject: [PATCH 11/90] fix(pabcd-state): honor project and environment hook switch (#252) --- .../components/pabcd-state/src/cli.ts | 23 ++++- .../components/pabcd-state/src/goal-gate.ts | 13 +-- .../components/pabcd-state/src/hook.ts | 2 + .../pabcd-state/src/interview-policy.ts | 14 +++ .../pabcd-state/test/goal-gate.test.ts | 23 +++++ .../pabcd-state/test/interview-policy.test.ts | 25 +++++- .../test/subagent-evidence.test.ts | 20 +++++ plugins/codexclaw/test/hook-e2e.test.mjs | 85 +++++++++++++++++++ 8 files changed, 197 insertions(+), 8 deletions(-) diff --git a/plugins/codexclaw/components/pabcd-state/src/cli.ts b/plugins/codexclaw/components/pabcd-state/src/cli.ts index b2f4f65f..90cbaa9c 100644 --- a/plugins/codexclaw/components/pabcd-state/src/cli.ts +++ b/plugins/codexclaw/components/pabcd-state/src/cli.ts @@ -51,6 +51,14 @@ import { handleIdleEditAdvisory } from "./idle-edit.ts"; import { handleMemoryWriteGate } from "./memory-write-gate.ts"; import { handleAutomationOwnershipGate } from "./automation-ownership-gate.ts"; import { handleReviewObserver } from "./review-observer.ts"; +import { readPabcdEnabled } from "./interview-policy.ts"; + +const PABCD_DISABLED_EVENTS = new Set([ + "session-start", "stop", "post-compact", "post-tool-use", + "subagent-stop", "subagent-stop-review", "pre-tool-use-idle-edit", + "pre-tool-use-friction", "post-tool-use-friction", "post-tool-use-edit-shape", + "post-tool-use-render-observation", +]); // wp10 (090 trim 4c): the ten terminal-only verb modules below are loaded with // dynamic import() inside their own branch instead of at module scope. @@ -392,6 +400,17 @@ async function main(): Promise { process.exit(0); } + let hookCwd = process.cwd(); + try { + const payload: unknown = JSON.parse(raw); + if (payload && typeof payload === "object" && !Array.isArray(payload)) { + const candidate = (payload as Record).cwd; + if (typeof candidate === "string" && candidate.length > 0) hookCwd = candidate; + } + } catch { /* malformed hook input keeps process cwd */ } + const pabcdEnabled = readPabcdEnabled(hookCwd); + if (!pabcdEnabled && PABCD_DISABLED_EVENTS.has(event)) process.exit(0); + // pre-tool-use is handled by a dedicated FAIL-CLOSED dispatcher: a thrown // error on a request_user_input call must DENY (R-9), never fail open. It is // outside the generic fail-open try below so the swallow cannot reopen the @@ -409,7 +428,7 @@ async function main(): Promise { if (payload) output = handleSessionStart(payload); // side-effect only; always "" } else if (event === "user-prompt-submit") { const payload = parseUserPromptSubmit(raw); - if (payload) output = handleUserPromptSubmit(payload); + if (payload) output = handleUserPromptSubmit(payload, process.platform, {}, { pabcdEnabled }); } else if (event === "stop") { const payload = parseStop(raw); if (payload) output = handleStop(payload); @@ -439,7 +458,7 @@ async function main(): Promise { // lint (deny-capable) first; a lint deny wins; otherwise the IDLE-edit advisory // may inject context. Both legs FAIL-OPEN; a crash must never deny the edit. output = handleApplyPatchLint(raw); - if (output === "") output = handleIdleEditAdvisory(raw); + if (pabcdEnabled && output === "") output = handleIdleEditAdvisory(raw); } else if (event === "pre-tool-use-idle-edit") { // 260714 wp3: FAIL-OPEN IDLE-edit advisory (IDLE-EDIT-ADVISORY-01). Allow + // additionalContext only; a crash here must never deny an edit. diff --git a/plugins/codexclaw/components/pabcd-state/src/goal-gate.ts b/plugins/codexclaw/components/pabcd-state/src/goal-gate.ts index 644e5de5..b5d93695 100644 --- a/plugins/codexclaw/components/pabcd-state/src/goal-gate.ts +++ b/plugins/codexclaw/components/pabcd-state/src/goal-gate.ts @@ -39,6 +39,7 @@ import { captureSessionSourceIdentity } from "./session-source-identity.ts"; import { resolveSessionSource } from "./session-source.ts"; import { parseSourceBoundReceipt } from "./source-receipt.ts"; import { hasSpentBudget, unrecordableVerdictStatus } from "./subagent-evidence.ts"; +import { readPabcdEnabled } from "./interview-policy.ts"; // Cross-component dist import (precedent: messenger-bridge/src/api-compat.ts:17). // 260724 WP1: deny remedies name `cxc orchestrate ...`/`cxc loop validate` — on a // payload-only install those must render the resolvable invocation. Emit-time only. @@ -206,7 +207,7 @@ function goalCompleteDenyEnvelope(reason: string): string { * Forensics: sessions 019f4407 (goal completed with a self-listed REMAINING queue) and * 019f4456 (empty goalplan would have rubber-stamped validate). */ -export function applyGoalCompleteGuard(payload: PreToolUsePayload): string { +export function applyGoalCompleteGuard(payload: PreToolUsePayload, pabcdEnabled = true): string { try { if (payload.hook_event_name !== "PreToolUse") return ""; if (payload.tool_name !== UPDATE_GOAL_TOOL_NAME) return ""; @@ -223,7 +224,7 @@ export function applyGoalCompleteGuard(payload: PreToolUsePayload): string { `GOAL-COMPLETE-GATE-01: this session's state is unreadable, so unresolved subagent evidence failures cannot be ruled out. Restore or reset the session state after verifying the delegated work, or use update_goal status "blocked".`, ); } - if (state.orchestrationActive && state.phase !== "IDLE" && state.phase !== "I") { + if (pabcdEnabled && state.orchestrationActive && state.phase !== "IDLE" && state.phase !== "I") { return goalCompleteDenyEnvelope( `GOAL-COMPLETE-GATE-01: a PABCD cycle is in flight at phase ${state.phase}. Close the cycle first (advance to D via \`cxc orchestrate ... --session ${payload.session_id}\`, or \`cxc orchestrate reset --session ${payload.session_id}\`), then mark the goal complete. If an external blocker prevents closing, use update_goal status "blocked" instead.`, ); @@ -269,7 +270,7 @@ export function applyGoalCompleteGuard(payload: PreToolUsePayload): string { `GOAL-COMPLETE-GATE-01: a delegated subagent exhausted its evidence-verification budget without a valid receipt. Re-verify that work and record a receipt with \`cxc evidence resolve --session ${payload.session_id} --agent --receipt \`, or use update_goal status "blocked".`, ); } - if (state.slug) { + if (pabcdEnabled && state.slug) { try { resolveSessionSource(payload.cwd, payload.session_id); } catch (err) { return goalCompleteDenyEnvelope(`SOURCE-ROOT: ${err instanceof Error ? err.message : String(err)}`); } const plan = readGoalplan(payload.cwd, state.slug); @@ -314,9 +315,11 @@ export function handlePreToolUseFailClosed(raw: string, deps: GoalActiveDeps = { try { const payload = parsePreToolUse(raw); if (!payload) return ""; + const enabled = readPabcdEnabled(payload.cwd); // Each guard is tool-name-scoped, so at most one fires. - return applyGoalBudgetGuard(payload) || applyGoalModeInterviewGuard(payload, deps) || applyGoalCompleteGuard(payload); + return applyGoalBudgetGuard(payload) || (enabled ? applyGoalModeInterviewGuard(payload, deps) : "") || applyGoalCompleteGuard(payload, enabled); } catch { - return rawLooksLikeRequestUserInput(raw) ? goalModeInterviewDenyEnvelope("unreadable") : ""; + return readPabcdEnabled(process.cwd()) && rawLooksLikeRequestUserInput(raw) + ? goalModeInterviewDenyEnvelope("unreadable") : ""; } } diff --git a/plugins/codexclaw/components/pabcd-state/src/hook.ts b/plugins/codexclaw/components/pabcd-state/src/hook.ts index 537df632..7d4987c7 100644 --- a/plugins/codexclaw/components/pabcd-state/src/hook.ts +++ b/plugins/codexclaw/components/pabcd-state/src/hook.ts @@ -625,6 +625,7 @@ export function handleUserPromptSubmit( payload: UserPromptSubmitPayload, platform: NodeJS.Platform = process.platform, dcloseCommitHooks: HookDcloseCommitHooks = {}, + options: { pabcdEnabled?: boolean } = {}, ): string { if (payload.hook_event_name !== "UserPromptSubmit") return ""; const turn = payload.turn_id ?? ""; @@ -649,6 +650,7 @@ export function handleUserPromptSubmit( // `cxc memory allow-write`; it must never break prompt handling. } } + if (options.pabcdEnabled === false) return ""; const state = readState(payload.cwd, payload.session_id); if (turn && state.injectedTurns.includes(turn)) return ""; diff --git a/plugins/codexclaw/components/pabcd-state/src/interview-policy.ts b/plugins/codexclaw/components/pabcd-state/src/interview-policy.ts index c4b8f776..15d1a4c0 100644 --- a/plugins/codexclaw/components/pabcd-state/src/interview-policy.ts +++ b/plugins/codexclaw/components/pabcd-state/src/interview-policy.ts @@ -43,6 +43,20 @@ export function configPath(cwd: string): string { return join(cwd, CONFIG_FILENAME); } +/** PABCD hook policy: a recognized environment value overrides project config. */ +export function readPabcdEnabled(cwd: string, env: NodeJS.ProcessEnv = process.env): boolean { + const override = env.CODEXCLAW_PABCD?.trim().toLowerCase(); + if (override === "off" || override === "0" || override === "false") return false; + if (override === "on" || override === "1" || override === "true") return true; + try { + const raw: unknown = JSON.parse(readFileSync(configPath(cwd), "utf8")); + if (!raw || typeof raw !== "object" || Array.isArray(raw)) return true; + const pabcd = (raw as Record).pabcd; + if (!pabcd || typeof pabcd !== "object" || Array.isArray(pabcd)) return true; + return (pabcd as Record).enabled !== false; + } catch { return true; } +} + /** * Read the policy for this repo. Missing file, unreadable file, malformed JSON and * unknown values all fall back to the default: a hook must never throw on a prompt. diff --git a/plugins/codexclaw/components/pabcd-state/test/goal-gate.test.ts b/plugins/codexclaw/components/pabcd-state/test/goal-gate.test.ts index 77e3e60a..631f3edb 100644 --- a/plugins/codexclaw/components/pabcd-state/test/goal-gate.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/goal-gate.test.ts @@ -456,3 +456,26 @@ test("GOAL-COMPLETE-GATE-01: fires through the fail-closed dispatcher", () => { assert.match(JSON.parse(out.trimEnd()).hookSpecificOutput.permissionDecisionReason, /GOAL-COMPLETE-GATE-01/); } finally { rmSync(cwd, { recursive: true, force: true }); } }); + +test("#252: PABCD off allows request_user_input but retains budget and evidence denials", () => { + const cwd = freshGateCwd(); + try { + writeFileSync(join(cwd, "codexclaw.json"), '{"pabcd":{"enabled":false}}'); + const raw = (tool_name: string, tool_input: unknown) => JSON.stringify({ + hook_event_name: "PreToolUse", session_id: "switch", cwd, tool_name, tool_input, + }); + assert.equal(handlePreToolUseFailClosed(raw("request_user_input", {}), depsWithStatus("active")), ""); + assert.equal(handlePreToolUseFailClosed(raw("request_user_input", {}), depsWithStatus("complete")), ""); + assert.match(handlePreToolUseFailClosed(raw("create_goal", { objective: "x", token_budget: 1 })), /token_budget/); + writeState(cwd, { ...defaultState("switch"), phase: "B", orchestrationActive: true }); + assert.equal(handlePreToolUseFailClosed(raw("update_goal", { status: "complete" })), ""); + const plan = buildGoalplan({ objective: "still open", criteria: [{ scenario: "test", expectedEvidence: "green" }] }); + writeGoalplan(cwd, plan); + writeState(cwd, { ...defaultState("switch"), slug: plan.slug }); + assert.equal(handlePreToolUseFailClosed(raw("update_goal", { status: "complete" })), ""); + writeState(cwd, { ...defaultState("switch"), phase: "B", orchestrationActive: true, + unverifiedSubagents: [{ agentId: "a1", turnId: "t1", agentType: "worker", attempts: 3, + receiptClaimed: "", recordedAt: "2026-09-28T00:00:00Z", resolvable: true }] }); + assert.match(handlePreToolUseFailClosed(raw("update_goal", { status: "complete" })), /unverified|evidence verification/); + } finally { rmSync(cwd, { recursive: true, force: true }); } +}); diff --git a/plugins/codexclaw/components/pabcd-state/test/interview-policy.test.ts b/plugins/codexclaw/components/pabcd-state/test/interview-policy.test.ts index eab4b1d3..e6f469d8 100644 --- a/plugins/codexclaw/components/pabcd-state/test/interview-policy.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/interview-policy.test.ts @@ -9,6 +9,7 @@ import { decideInterviewEntry, isInterviewPolicy, readInterviewPolicy, + readPabcdEnabled, writeInterviewPolicy, type InterviewPolicy, } from "../src/interview-policy.ts"; @@ -128,12 +129,34 @@ test("wp5: writing the policy round-trips through the reader", () => { }); test("wp5: writing preserves unrelated keys in codexclaw.json", () => { - const dir = repoWith(JSON.stringify({ somethingElse: { nested: 1 }, interview: "off" })); + const dir = repoWith(JSON.stringify({ somethingElse: { nested: 1 }, interview: "off", pabcd: { enabled: false } })); const res = writeInterviewPolicy(dir, "new-unit"); assert.ok(res.ok); const parsed = JSON.parse(readFileSync(join(dir, CONFIG_FILENAME), "utf8")); assert.deepEqual(parsed.somethingElse, { nested: 1 }, "a foreign key must survive"); assert.equal(parsed.interview, "new-unit"); + assert.deepEqual(parsed.pabcd, { enabled: false }); +}); + +test("#252: PABCD environment override and project fallback matrix", () => { + const disabled = repoWith('{"pabcd":{"enabled":false}}'); + const enabled = repoWith('{"pabcd":{"enabled":true}}'); + const absent = repoWith(null); + for (const value of ["off", " OFF ", "0", "false", " FALSE "]) { + for (const cwd of [enabled, absent]) assert.equal(readPabcdEnabled(cwd, { CODEXCLAW_PABCD: value }), false, value); + } + for (const value of ["on", " ON ", "1", "true", " TRUE "]) { + assert.equal(readPabcdEnabled(disabled, { CODEXCLAW_PABCD: value }), true, value); + } + for (const value of ["", "other", undefined]) { + const env = { CODEXCLAW_PABCD: value }; + assert.equal(readPabcdEnabled(disabled, env), false); + assert.equal(readPabcdEnabled(enabled, env), true); + assert.equal(readPabcdEnabled(absent, env), true); + } + for (const contents of ["{ broken", "[]", "null", '{"pabcd":[]}', '{"pabcd":{"enabled":"false"}}']) { + assert.equal(readPabcdEnabled(repoWith(contents), {}), true, contents); + } }); test("wp5: a malformed file is replaced and the caller is told", () => { diff --git a/plugins/codexclaw/components/pabcd-state/test/subagent-evidence.test.ts b/plugins/codexclaw/components/pabcd-state/test/subagent-evidence.test.ts index 27c7f8b7..d7d574a7 100644 --- a/plugins/codexclaw/components/pabcd-state/test/subagent-evidence.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/subagent-evidence.test.ts @@ -5,6 +5,7 @@ * valid-receipt release, symlink/outside-root rejection, and transcript spoofing. */ import { test } from "node:test"; +import { spawnSync } from "node:child_process"; import assert from "node:assert/strict"; import { mkdtempSync, mkdirSync, writeFileSync, symlinkSync, existsSync, chmodSync, rmSync, readdirSync } from "node:fs"; import { tmpdir } from "node:os"; @@ -823,3 +824,22 @@ test("canonical executor exit without a receipt is blocked", () => { const out = runSubagentStopGate(payload(cwd, { agent_type: "executor" })); assert.equal(JSON.parse(out).decision, "block"); }); + +test("#252: policy-off dispatch leaves armed executor and worker evidence untouched", () => { + const cwd = tmp(); + try { + writeFileSync(join(cwd, "codexclaw.json"), '{"pabcd":{"enabled":false}}'); + writeState(cwd, { ...defaultState("s1"), phase: "B", orchestrationActive: true }); + const entry = new URL("../src/cli.ts", import.meta.url); + for (const agent_type of ["executor", "worker"]) { + const result = spawnSync(process.execPath, [entry.pathname, "hook", "subagent-stop"], { + input: JSON.stringify(payload(cwd, { agent_type, agent_id: agent_type })), encoding: "utf8", + env: { ...process.env, CODEXCLAW_PABCD: "off" }, + }); + assert.equal(result.status, 0, result.stderr); + assert.equal(result.stdout, ""); + assert.equal(readAttempts(cwd, "s1", agent_type), 0); + } + assert.deepEqual(readState(cwd, "s1").unverifiedSubagents, []); + } finally { rmSync(cwd, { recursive: true, force: true }); } +}); diff --git a/plugins/codexclaw/test/hook-e2e.test.mjs b/plugins/codexclaw/test/hook-e2e.test.mjs index ce3c68e0..3a2b52fe 100644 --- a/plugins/codexclaw/test/hook-e2e.test.mjs +++ b/plugins/codexclaw/test/hook-e2e.test.mjs @@ -132,6 +132,91 @@ function emptyCodexHome() { return { dir, env: { CODEX_HOME: dir, CODEXCLAW_HOME: join(dir, "cxc"), CODEX_SQLITE_HOME: dir } }; } +test("#252: PABCD off silences component hooks and both SubagentStop roles", () => { + const { distAbs } = readHookCommand("./hooks/session-start-bootstrapping-pabcd-state.json"); + const ep = snapshotEntrypoint(distAbs); + assert.ok(ep); + const cwd = mkdtempSync(join(tmpdir(), "ccx-pabcd-off-")); + try { + writeFileSync(join(cwd, "codexclaw.json"), '{"pabcd":{"enabled":false}}'); + for (const [event, hook_event_name] of [ + ["session-start", "SessionStart"], ["stop", "Stop"], ["post-compact", "PostCompact"], + ["subagent-stop-review", "SubagentStop"], + ]) { + const result = runHook(ep, event, { hook_event_name, session_id: "off", cwd }); + assert.equal(result.status, 0, result.stderr); + assert.equal(result.stdout, "", event); + } + assert.equal(existsSync(join(cwd, ".codexclaw", "sessions", "off.json")), false); + mkdirSync(join(cwd, ".codexclaw", "sessions"), { recursive: true }); + writeFileSync(join(cwd, ".codexclaw", "sessions", "off.json"), + JSON.stringify({ sessionId: "off", phase: "B", orchestrationActive: true })); + for (const agent_type of ["executor", "worker"]) { + const result = runHook(ep, "subagent-stop", { hook_event_name: "SubagentStop", session_id: "off", cwd, + agent_type, agent_id: agent_type, last_assistant_message: "done" }); + assert.equal(result.status, 0, result.stderr); + assert.equal(result.stdout, "", agent_type); + assert.equal(existsSync(join(cwd, ".codexclaw", "evidence-attempts")), false); + } + const prompt = runHook(ep, "user-prompt-submit", { hook_event_name: "UserPromptSubmit", session_id: "off", cwd, + turn_id: "t1", prompt: "Plan this feature" }); + assert.equal(prompt.status, 0, prompt.stderr); + assert.equal(prompt.stdout, ""); + } finally { rmSync(cwd, { recursive: true, force: true }); } +}); + +test("#252: env on overrides project false and gates executor", () => { + const { distAbs } = readHookCommand("./hooks/subagent-stop-verifying-evidence.json"); + const ep = snapshotEntrypoint(distAbs); + assert.ok(ep); + const cwd = mkdtempSync(join(tmpdir(), "ccx-pabcd-on-")); + try { + writeFileSync(join(cwd, "codexclaw.json"), '{"pabcd":{"enabled":false}}'); + const result = runHook(ep, "subagent-stop", { hook_event_name: "SubagentStop", session_id: "on", cwd, + agent_type: "executor", agent_id: "a1", last_assistant_message: "done" }, { CODEXCLAW_PABCD: " ON " }); + assert.equal(result.status, 0, result.stderr); + assert.equal(JSON.parse(result.stdout).decision, "block"); + } finally { rmSync(cwd, { recursive: true, force: true }); } +}); + +test("#252: PABCD off retains memory marker and independent lint and automation guards", () => { + const { distAbs } = readHookCommand("./hooks/user-prompt-submit-checking-pabcd-trigger.json"); + const ep = snapshotEntrypoint(distAbs); + assert.ok(ep); + const cwd = mkdtempSync(join(tmpdir(), "ccx-pabcd-memory-")); + const env = { CODEXCLAW_PABCD: "off" }; + try { + const prompt = (session_id, turn_id, text) => runHook(ep, "user-prompt-submit", { + hook_event_name: "UserPromptSubmit", session_id, cwd, turn_id, prompt: text }, env); + assert.equal(prompt("remember", "t1", "Remember this: test marker").stdout, ""); + const input = (session_id) => ({ hook_event_name: "PreToolUse", session_id, cwd, + tool_name: "memoriesadd_ad_hoc_note", tool_input: { filename: "note.md", note: "x" } }); + const allowed = runHook(ep, "pre-tool-use-memory-write", input("remember"), env); + assert.equal(allowed.stdout, ""); + assert.equal(prompt("ordinary", "t1", "Plan this feature").stdout, ""); + const denied = runHook(ep, "pre-tool-use-memory-write", input("ordinary"), env); + assert.equal(JSON.parse(denied.stdout).hookSpecificOutput.permissionDecision, "deny"); + const lint = runHook(ep, "pre-tool-use-edit", { hook_event_name: "PreToolUse", session_id: "ordinary", cwd, + tool_name: "apply_patch", tool_input: { command: "+++ b/x.ts\n+const x = foo as any;\n" } }, env); + assert.equal(JSON.parse(lint.stdout).hookSpecificOutput.permissionDecision, "deny"); + const clean = runHook(ep, "pre-tool-use-edit", { hook_event_name: "PreToolUse", session_id: "ordinary", cwd, + tool_name: "apply_patch", tool_input: { command: "+++ b/x.ts\n+const x: number = 1;\n" } }, env); + assert.equal(clean.stdout, "", "PABCD idle-edit advisory stays silent"); + const automation = runHook(ep, "pre-tool-use-automation-ownership", { hook_event_name: "PreToolUse", + session_id: "ordinary", cwd, tool_name: "mcp__codex_app__automation_update", + tool_input: { mode: "delete", id: "foreign" } }, env); + assert.equal(JSON.parse(automation.stdout).hookSpecificOutput.permissionDecision, "deny"); + const managed = join(cwd, "worktrees", "slot", "repo"); + mkdirSync(managed, { recursive: true }); + writeFileSync(join(managed, ".git"), "gitdir: /fake/worktree\n"); + const worktree = runHook(ep, "worktree-guard-pretool", { hook_event_name: "PreToolUse", + session_id: "ordinary", cwd: managed, tool_name: "Bash", + tool_input: { command: `git worktree remove ${managed}` } }, + { ...env, CODEX_HOME: cwd }); + assert.equal(JSON.parse(worktree.stdout).hookSpecificOutput.permissionDecision, "deny"); + } finally { rmSync(cwd, { recursive: true, force: true }); } +}); + test("WP7/G19: every manifest hook command resolves to an existing dist entrypoint", () => { const manifest = JSON.parse(readFileSync(manifestPath, "utf8")); // 260804: 18 -> 21 with the worktree-guard hooks (session-start-detecting- From ca6727ceed321ead2816b345d6369b2b6a2776a8 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 00:34:04 +0900 Subject: [PATCH 12/90] fix(pabcd-state): gate worker receipts only in active B/C (#251) --- .../pabcd-state/src/subagent-evidence.ts | 29 +++- .../test/subagent-evidence.test.ts | 136 +++++++++++++++--- plugins/codexclaw/test/hook-e2e.test.mjs | 38 ++++- 3 files changed, 183 insertions(+), 20 deletions(-) diff --git a/plugins/codexclaw/components/pabcd-state/src/subagent-evidence.ts b/plugins/codexclaw/components/pabcd-state/src/subagent-evidence.ts index ac91f52a..2eb75838 100644 --- a/plugins/codexclaw/components/pabcd-state/src/subagent-evidence.ts +++ b/plugins/codexclaw/components/pabcd-state/src/subagent-evidence.ts @@ -1,7 +1,7 @@ /** * subagent-evidence.ts — SubagentStop evidence-receipt gate (lazygap_impl 010). * - * A dispatched WRITE/verify subagent (agent_type "executor", or legacy "worker") cannot "finish" without a + * A registered executor, or a legacy worker in an active PABCD B/C cycle, cannot "finish" without a * non-empty evidence receipt under `.codexclaw/evidence/`. Missing/invalid receipt -> * `decision:"block"` with a verifier directive that re-prompts the CHILD (codex-rs * turn.rs:323). After MAX_ATTEMPTS the directive escalates but remains fail-closed; @@ -53,10 +53,11 @@ import { type UnverifiedSubagent, } from "./state.ts"; import type { SubagentStopPayload } from "./hook.ts"; +import { configPath } from "./interview-policy.ts"; /** - * agent_type values this gate refuses to release without a receipt. - * DISPATCH-AGENT-TYPE-01: executor and legacy worker are gated. Read-only audit/research + * agent_type values routed to this gate. + * DISPATCH-AGENT-TYPE-01: executor and legacy worker are candidates. Read-only audit/research * dispatches MUST use agent_type:"explorer" so they bypass both the hook * manifest matcher (^(executor|worker)$) and this runtime gate. See * structure/20_pabcd_dispatch_doctrine.md §3. @@ -468,9 +469,31 @@ export function escalationDirective(): string { * The SubagentStop decision. Returns the codex hook stdout (a `{decision:"block",reason}` * JSON string to force the child to continue, or `""` to release). Total: never throws. */ +function readPabcdEnabled(cwd: string): boolean { + // WP2-012 owns the shared reader. Keep this isolated branch buildable until its + // predecessor lands; the policy precedence and shape match that reader. + const override = process.env.CODEXCLAW_PABCD?.trim().toLowerCase(); + if (override === "off" || override === "0" || override === "false") return false; + if (override === "on" || override === "1" || override === "true") return true; + try { + const raw: unknown = JSON.parse(readFileSync(configPath(cwd), "utf8")); + if (!raw || typeof raw !== "object" || Array.isArray(raw)) return true; + const pabcd = (raw as Record).pabcd; + if (!pabcd || typeof pabcd !== "object" || Array.isArray(pabcd)) return true; + return (pabcd as Record).enabled !== false; + } catch { + return true; + } +} + export function runSubagentStopGate(payload: SubagentStopPayload): string { try { + if (!readPabcdEnabled(payload.cwd)) return ""; if (!GATED_AGENT_TYPES.has(payload.agent_type)) return ""; + if (payload.agent_type === "worker") { + const { state, unreadable } = readStateStrict(payload.cwd, payload.session_id); + if (unreadable || !state.orchestrationActive || (state.phase !== "B" && state.phase !== "C")) return ""; + } const agentId = payload.agent_id ?? ""; const { cwd, session_id: sessionId } = payload; // Same identity as a tombstone: two turns of one agent must not share a budget. diff --git a/plugins/codexclaw/components/pabcd-state/test/subagent-evidence.test.ts b/plugins/codexclaw/components/pabcd-state/test/subagent-evidence.test.ts index 27c7f8b7..e9dbb67c 100644 --- a/plugins/codexclaw/components/pabcd-state/test/subagent-evidence.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/subagent-evidence.test.ts @@ -6,7 +6,7 @@ */ import { test } from "node:test"; import assert from "node:assert/strict"; -import { mkdtempSync, mkdirSync, writeFileSync, symlinkSync, existsSync, chmodSync, rmSync, readdirSync } from "node:fs"; +import { mkdtempSync, mkdirSync, readFileSync, writeFileSync, symlinkSync, existsSync, chmodSync, rmSync, readdirSync } from "node:fs"; import { tmpdir } from "node:os"; import { dirname, join } from "node:path"; import { fileURLToPath, pathToFileURL } from "node:url"; @@ -34,7 +34,7 @@ function payload(cwd: string, over: Partial = {}): Subagent hook_event_name: "SubagentStop", session_id: "s1", cwd, - agent_type: "worker", + agent_type: "executor", agent_id: "a1", last_assistant_message: null, ...over, @@ -54,22 +54,122 @@ test("010: non-gated agent_type (explorer) is released untouched", () => { assert.equal(out, ""); }); -test("010: worker with no receipt blocks (under cap) and names the receipt contract", () => { +test("251: worker outside armed PABCD releases without attempts or tombstone", () => { const cwd = tmp(); - const out = runSubagentStopGate(payload(cwd)); - const parsed = JSON.parse(out); - assert.equal(parsed.decision, "block"); - assert.match(parsed.reason, /EVIDENCE_RECORDED/); - assert.equal(readAttempts(cwd, "s1", "a1"), 1); + assert.equal(runSubagentStopGate(payload(cwd, { agent_type: "worker" })), ""); + assert.equal(readAttempts(cwd, "s1", "a1"), 0); + assert.equal(existsSync(join(cwd, ".codexclaw", "evidence-attempts")), false); + assert.deepEqual(readState(cwd, "s1").unverifiedSubagents, []); + assert.equal(existsSync(join(cwd, ".codexclaw")), false); +}); + +test("251: worker in armed B and C blocks without receipt", () => { + for (const phase of ["B", "C"] as const) { + const cwd = tmp(); + writeState(cwd, { ...defaultState("s1"), phase, orchestrationActive: true }); + const out = runSubagentStopGate(payload(cwd, { agent_type: "worker" })); + assert.equal(JSON.parse(out).decision, "block"); + assert.equal(readAttempts(cwd, "s1", "a1"), 1); + } +}); + +test("251: worker at P/A/IDLE or inactive B/C releases without writes", () => { + for (const phase of ["P", "A", "IDLE", "B", "C"] as const) { + const cwd = tmp(); + writeState(cwd, { ...defaultState("s1"), phase, orchestrationActive: phase !== "B" && phase !== "C" }); + assert.equal(runSubagentStopGate(payload(cwd, { agent_type: "worker" })), "", phase); + assert.equal(readAttempts(cwd, "s1", "a1"), 0); + assert.equal(existsSync(join(cwd, ".codexclaw", "evidence-attempts")), false); + assert.deepEqual(readState(cwd, "s1").unverifiedSubagents, []); + } +}); + +test("251: worker with unreadable state releases without write", () => { + const cwd = tmp(); + const statePath = join(cwd, ".codexclaw", "sessions", "s1.json"); + mkdirSync(dirname(statePath), { recursive: true }); + writeFileSync(statePath, "{ corrupt"); + assert.equal(runSubagentStopGate(payload(cwd, { agent_type: "worker" })), ""); + assert.equal(readAttempts(cwd, "s1", "a1"), 0); + assert.equal(existsSync(join(cwd, ".codexclaw", "evidence-attempts")), false); + assert.equal(readFileSync(statePath, "utf8"), "{ corrupt"); +}); + +test("251: executor with or without state blocks outside an active cycle", () => { + for (const present of [false, true]) { + const cwd = tmp(); + if (present) writeState(cwd, defaultState("s1")); + assert.equal(JSON.parse(runSubagentStopGate(payload(cwd, { agent_type: "executor" }))).decision, "block"); + assert.equal(readAttempts(cwd, "s1", "a1"), 1); + } +}); + +test("251: disabled PABCD releases executor and worker with no attempts or tombstones", () => { + const prior = process.env.CODEXCLAW_PABCD; + process.env.CODEXCLAW_PABCD = "off"; + try { + for (const agent_type of ["executor", "worker"]) { + const cwd = tmp(); + writeState(cwd, { ...defaultState("s1"), phase: "B", orchestrationActive: true }); + assert.equal(runSubagentStopGate(payload(cwd, { agent_type })), ""); + assert.equal(readAttempts(cwd, "s1", "a1"), 0); + assert.equal(existsSync(join(cwd, ".codexclaw", "evidence-attempts")), false); + assert.deepEqual(readState(cwd, "s1").unverifiedSubagents, []); + } + } finally { + if (prior === undefined) delete process.env.CODEXCLAW_PABCD; + else process.env.CODEXCLAW_PABCD = prior; + } +}); + +test("251: project-disabled PABCD releases both roles in an armed cycle", () => { + const cwd = tmp(); + writeFileSync(join(cwd, "codexclaw.json"), JSON.stringify({ pabcd: { enabled: false } })); + writeState(cwd, { ...defaultState("s1"), phase: "C", orchestrationActive: true }); + for (const agent_type of ["executor", "worker"]) { + assert.equal(runSubagentStopGate(payload(cwd, { agent_type })), ""); + assert.equal(readAttempts(cwd, "s1", "a1"), 0); + } + assert.equal(existsSync(join(cwd, ".codexclaw", "evidence-attempts")), false); + assert.deepEqual(readState(cwd, "s1").unverifiedSubagents, []); +}); + +test("251: armed worker rejects invalid receipt and records terminal verdict", () => { + const cwd = tmp(); + writeState(cwd, { ...defaultState("s1"), phase: "B", orchestrationActive: true }); + for (let i = 0; i < MAX_ATTEMPTS; i++) { + const out = runSubagentStopGate(payload(cwd, { + agent_type: "worker", last_assistant_message: "EVIDENCE_RECORDED: elsewhere/missing.md", + })); + assert.equal(JSON.parse(out).decision, "block"); + } + assert.equal(runSubagentStopGate(payload(cwd, { agent_type: "worker" })), ""); + assert.equal(readState(cwd, "s1").unverifiedSubagents[0]?.agentType, "worker"); + assert.equal(readAttempts(cwd, "s1", "a1"), MAX_ATTEMPTS); +}); + +test("251: enabled PABCD overrides project false and gates executor", () => { + const cwd = tmp(); + writeFileSync(join(cwd, "codexclaw.json"), JSON.stringify({ pabcd: { enabled: false } })); + const prior = process.env.CODEXCLAW_PABCD; + process.env.CODEXCLAW_PABCD = "on"; + try { + assert.equal(JSON.parse(runSubagentStopGate(payload(cwd, { agent_type: "executor" }))).decision, "block"); + assert.equal(readAttempts(cwd, "s1", "a1"), 1); + } finally { + if (prior === undefined) delete process.env.CODEXCLAW_PABCD; + else process.env.CODEXCLAW_PABCD = prior; + } }); test("010: worker with a valid receipt is released and attempts cleared", () => { const cwd = tmp(); + writeState(cwd, { ...defaultState("s1"), phase: "B", orchestrationActive: true }); writeEvidence(cwd, "proof.md", "ran tests: 369/369"); // prime an attempt to prove it gets cleared on success. - runSubagentStopGate(payload(cwd)); + runSubagentStopGate(payload(cwd, { agent_type: "worker" })); const out = runSubagentStopGate( - payload(cwd, { last_assistant_message: "done.\nEVIDENCE_RECORDED: .codexclaw/evidence/proof.md" }), + payload(cwd, { agent_type: "worker", last_assistant_message: "done.\nEVIDENCE_RECORDED: .codexclaw/evidence/proof.md" }), ); assert.equal(out, ""); assert.equal(readAttempts(cwd, "s1", "a1"), 0); @@ -151,7 +251,7 @@ test("010: terminal release records an unresolved tombstone for the parent", () assert.equal(state.unverifiedSubagents.length, 1); const entry = state.unverifiedSubagents[0]; assert.equal(entry.agentId, "a1"); - assert.equal(entry.agentType, "worker"); + assert.equal(entry.agentType, "executor"); assert.equal(entry.resolvable, true); }); @@ -264,7 +364,7 @@ test("010: concurrent terminal stops do not lose a tombstone", async () => { for (let i = 0; i < MAX_ATTEMPTS; i++) runSubagentStopGate(payload(cwd, { agent_id: agent })); } const runner = (agent: string) => - `import{runSubagentStopGate}from${JSON.stringify(src)};runSubagentStopGate({hook_event_name:"SubagentStop",session_id:"s1",cwd:${JSON.stringify(cwd)},agent_type:"worker",agent_id:${JSON.stringify(agent)},last_assistant_message:null});`; + `import{runSubagentStopGate}from${JSON.stringify(src)};runSubagentStopGate({hook_event_name:"SubagentStop",session_id:"s1",cwd:${JSON.stringify(cwd)},agent_type:"executor",agent_id:${JSON.stringify(agent)},last_assistant_message:null});`; const { spawn } = await import("node:child_process"); const childErrors: string[] = []; const go = (agent: string) => @@ -768,18 +868,20 @@ test("DISPATCH-AGENT-TYPE-01: default agent_type is not gated", () => { test("DISPATCH-AGENT-TYPE-01: worker cannot exempt itself with transcript text", () => { const cwd = tmp(); + writeState(cwd, { ...defaultState("s1"), phase: "B", orchestrationActive: true }); const transcriptDir = join(cwd, ".codex", "sessions"); mkdirSync(transcriptDir, { recursive: true }); const transcriptPath = join(transcriptDir, "child.jsonl"); writeFileSync(transcriptPath, '[CXC-EVIDENCE-EXEMPT] [REVIEWER] review the plan\n'); const out = runSubagentStopGate( - payload(cwd, { agent_transcript_path: transcriptPath }), + payload(cwd, { agent_type: "worker", agent_transcript_path: transcriptPath }), ); assert.equal(JSON.parse(out).decision, "block"); }); test("DISPATCH-AGENT-TYPE-01: marker deep in transcript still cannot bypass", () => { const cwd = tmp(); + writeState(cwd, { ...defaultState("s1"), phase: "B", orchestrationActive: true }); const transcriptDir = join(cwd, ".codex", "sessions"); mkdirSync(transcriptDir, { recursive: true }); const transcriptPath = join(transcriptDir, "child.jsonl"); @@ -787,19 +889,20 @@ test("DISPATCH-AGENT-TYPE-01: marker deep in transcript still cannot bypass", () const padding = "x".repeat(30000); writeFileSync(transcriptPath, padding + '\n[CXC-EVIDENCE-EXEMPT] task\n'); const out = runSubagentStopGate( - payload(cwd, { agent_transcript_path: transcriptPath }), + payload(cwd, { agent_type: "worker", agent_transcript_path: transcriptPath }), ); assert.equal(JSON.parse(out).decision, "block"); }); test("DISPATCH-AGENT-TYPE-01: generic read-only text without token still blocks", () => { const cwd = tmp(); + writeState(cwd, { ...defaultState("s1"), phase: "B", orchestrationActive: true }); const transcriptDir = join(cwd, ".codex", "sessions"); mkdirSync(transcriptDir, { recursive: true }); const transcriptPath = join(transcriptDir, "child.jsonl"); writeFileSync(transcriptPath, '[REVIEWER read-only] review the plan\n'); const out = runSubagentStopGate( - payload(cwd, { agent_transcript_path: transcriptPath }), + payload(cwd, { agent_type: "worker", agent_transcript_path: transcriptPath }), ); const parsed = JSON.parse(out); assert.equal(parsed.decision, "block", "generic read-only without token should still block"); @@ -807,12 +910,13 @@ test("DISPATCH-AGENT-TYPE-01: generic read-only text without token still blocks" test("DISPATCH-AGENT-TYPE-01: worker without token still blocks", () => { const cwd = tmp(); + writeState(cwd, { ...defaultState("s1"), phase: "B", orchestrationActive: true }); const transcriptDir = join(cwd, ".codex", "sessions"); mkdirSync(transcriptDir, { recursive: true }); const transcriptPath = join(transcriptDir, "child.jsonl"); writeFileSync(transcriptPath, 'TASK: implement the fix and write tests.\n'); const out = runSubagentStopGate( - payload(cwd, { agent_transcript_path: transcriptPath }), + payload(cwd, { agent_type: "worker", agent_transcript_path: transcriptPath }), ); const parsed = JSON.parse(out); assert.equal(parsed.decision, "block", "write task should still be gated"); diff --git a/plugins/codexclaw/test/hook-e2e.test.mjs b/plugins/codexclaw/test/hook-e2e.test.mjs index ce3c68e0..a86c607a 100644 --- a/plugins/codexclaw/test/hook-e2e.test.mjs +++ b/plugins/codexclaw/test/hook-e2e.test.mjs @@ -736,7 +736,12 @@ for (const agentType of ["executor", "worker"]) test(`L010: subagent-stop hook e if (!ep) return; const tmp = mkdtempSync(join(tmpdir(), "ccx-sas-")); try { - // 1) worker, no receipt -> block with the EVIDENCE_RECORDED contract. + // The legacy worker gate requires the parent's active B/C state. + if (agentType === "worker") { + mkdirSync(join(tmp, ".codexclaw", "sessions"), { recursive: true }); + for (const id of ["s1", "s3"]) writeFileSync(join(tmp, ".codexclaw", "sessions", `${id}.json`), JSON.stringify({ phase: "B", orchestrationActive: true })); + } + // 1) armed worker or executor, no receipt -> block with the EVIDENCE_RECORDED contract. const blocked = runHook(ep, hookEvent, { hook_event_name: "SubagentStop", session_id: "s1", cwd: tmp, agent_type: agentType, agent_id: "a1", last_assistant_message: "all done!", @@ -767,6 +772,37 @@ for (const agentType of ["executor", "worker"]) test(`L010: subagent-stop hook e } finally { rmSync(tmp, { recursive: true, force: true }); } }); +test("251: built SubagentStop releases unarmed worker and policy-disabled roles", () => { + const { hookEvent, distAbs } = readHookCommand("./hooks/subagent-stop-verifying-evidence.json"); + const ep = snapshotEntrypoint(distAbs); + assert.ok(ep, "built hook entrypoint required"); + const tmp = mkdtempSync(join(tmpdir(), "ccx-sas-251-")); + const run = (agent_type, env = {}) => runHook(ep, hookEvent, { + hook_event_name: "SubagentStop", session_id: "s1", cwd: tmp, + agent_type, agent_id: agent_type, last_assistant_message: null, + }, env); + try { + const free = run("worker"); + assert.equal(free.status, 0, free.stderr); + assert.equal(free.stdout, ""); + assert.equal(existsSync(join(tmp, ".codexclaw")), false); + + mkdirSync(join(tmp, ".codexclaw", "sessions"), { recursive: true }); + writeFileSync(join(tmp, ".codexclaw", "sessions", "s1.json"), JSON.stringify({ phase: "C", orchestrationActive: true })); + for (const role of ["executor", "worker"]) { + const disabled = run(role, { CODEXCLAW_PABCD: "off" }); + assert.equal(disabled.status, 0, disabled.stderr); + assert.equal(disabled.stdout, ""); + assert.equal(existsSync(join(tmp, ".codexclaw", "evidence-attempts")), false); + assert.deepEqual(JSON.parse(readFileSync(join(tmp, ".codexclaw", "sessions", "s1.json"), "utf8")), { phase: "C", orchestrationActive: true }); + } + writeFileSync(join(tmp, "codexclaw.json"), JSON.stringify({ pabcd: { enabled: false } })); + const enabled = run("executor", { CODEXCLAW_PABCD: "on" }); + assert.equal(enabled.status, 0, enabled.stderr); + assert.equal(JSON.parse(enabled.stdout).decision, "block"); + } finally { rmSync(tmp, { recursive: true, force: true }); } +}); + // 260710: the spawn hook repairs provided cxc mentions on both schemas without // inventing baselines. 260710 parity: BOTH surfaces get D1/D2 leaf guarding and // configured model/effort routing; V2 additionally gets SKILL.md body inlining. From 3ab86999330d2f640cc3266b18cda61bb6c941ee Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 00:34:04 +0900 Subject: [PATCH 13/90] docs(pabcd): explain hook policy switch (#252) --- docs-site/src/content/docs/guides/pabcd.md | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/docs-site/src/content/docs/guides/pabcd.md b/docs-site/src/content/docs/guides/pabcd.md index ac469ca8..2796314f 100644 --- a/docs-site/src/content/docs/guides/pabcd.md +++ b/docs-site/src/content/docs/guides/pabcd.md @@ -61,6 +61,18 @@ cxc orchestrate reset ## Stop continuation +### Disable PABCD hooks + +Set `CODEXCLAW_PABCD=off` (also accepts `0` or `false`), or add this to the project-root `codexclaw.json`: + +```json +{ "pabcd": { "enabled": false } } +``` + +The environment setting takes precedence: `on`, `1`, or `true` enables PABCD hooks even when the project setting is false. Values are case-insensitive and surrounding spaces are ignored. An unrecognized value uses the project setting. PABCD hooks are enabled by default when the setting is absent or the config is malformed. + +This switch silences PABCD hook dispatch in this component. Worktree, memory-write, automation-ownership, apply-patch lint, and independent goal safety guards remain active. It does not erase session state or disable CLI commands. `cxc config interview off` changes only Interview promotion; it does not disable PABCD hooks. + Under an **active native goal**, the `Stop` hook returns `{"decision":"block","reason":...}` to keep the agent advancing — both mid-cycle (continue the current phase) and at IDLE with no in-flight cycle From e08f8748c157ddfb401b3ecd9a8197f57919db98 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 00:34:17 +0900 Subject: [PATCH 14/90] docs(pabcd): describe executor and worker receipt scope (#251) --- .../components/subagent-config/src/spawn-wrapper.ts | 2 ++ .../codexclaw/skills/pabcd/references/delegation.md | 1 + structure/20_pabcd_dispatch_doctrine.md | 12 +++++++----- 3 files changed, 10 insertions(+), 5 deletions(-) diff --git a/plugins/codexclaw/components/subagent-config/src/spawn-wrapper.ts b/plugins/codexclaw/components/subagent-config/src/spawn-wrapper.ts index 8ddc217c..073f72f9 100644 --- a/plugins/codexclaw/components/subagent-config/src/spawn-wrapper.ts +++ b/plugins/codexclaw/components/subagent-config/src/spawn-wrapper.ts @@ -10,6 +10,8 @@ * explorer/reviewer -> "explorer", executor -> "worker". Architect never aliases another role. * executor resolves to its registered native "executor" type when $CODEX_HOME/agents/executor.toml * exists (cxc subagents register executor); unregistered installs keep built-in worker. + * With PABCD enabled, executor is receipt-gated on every stop; worker is gated only + * while the parent is actively orchestrating B/C. Both release when policy is off. * - the role prompt is injected INLINE in the message ("TASK: ..."), since plugin * install dirs are not a config layer. * - model selection is not emitted by the v2 builder. The durable per-role model in diff --git a/plugins/codexclaw/skills/pabcd/references/delegation.md b/plugins/codexclaw/skills/pabcd/references/delegation.md index 285963aa..848b4553 100644 --- a/plugins/codexclaw/skills/pabcd/references/delegation.md +++ b/plugins/codexclaw/skills/pabcd/references/delegation.md @@ -10,6 +10,7 @@ At P, consult a read-only architect; at A, dispatch an independent reviewer. Use a supported read-only transport for both and a supported write role for bounded implementation (DISPATCH-AGENT-TYPE-01 and the live schema below). The executor role resolves to its registered native `executor` type once `cxc subagents register executor` has run; unregistered installs keep the built-in `worker`. +When PABCD policy is enabled, the registered executor is evidence-gated on every SubagentStop; the built-in worker fallback is evidence-gated only while the parent has an active PABCD B/C cycle. When PABCD policy is disabled, both gates are silent. Outside that cycle the worker releases without a receipt. Subagents are leaves (LEAF-TOPOLOGY-01) unless recursion is explicitly granted. Every dispatch carries a structured TASK packet (DISPATCH-TASK-01): `TASK`, `SCOPE`, `MUST DO`, `MUST NOT`, `PROOF`, `RETURN FORMAT`, and decision boundary. diff --git a/structure/20_pabcd_dispatch_doctrine.md b/structure/20_pabcd_dispatch_doctrine.md index fa707bb4..a3d513b8 100644 --- a/structure/20_pabcd_dispatch_doctrine.md +++ b/structure/20_pabcd_dispatch_doctrine.md @@ -118,14 +118,16 @@ codexclaw translation: and fresh-session schema verification are prerequisites; never alias it to explorer or reviewer when unavailable. Registration is not a dispatch side effect. - **DISPATCH-AGENT-TYPE-01 (DEFAULT).** The role-to-agent-type mapping above is the - canonical dispatch classifier for the SubagentStop evidence gate: only - `agent_type:"worker"` triggers the evidence-receipt gate (hook matcher `^worker$` + - runtime `GATED_AGENT_TYPES`). Read-only audit, research, and review dispatches MUST + canonical dispatch classifier for the SubagentStop evidence gate: registered + `executor` is receipt-gated under enabled PABCD; the legacy `worker` is gated only + while the parent has an active B/C cycle (hook matcher `^(executor|worker)$` + + runtime `GATED_AGENT_TYPES`). Both gates are silent under disabled PABCD. + Read-only audit, research, and review dispatches MUST use a supported read-only role: a registered role when exposed, otherwise `agent_type:"explorer"` when available. With schema-minimal native tools, carry the logical role and read-only scope in the message instead; that - label is not native role enforcement and cannot evade an actual worker receipt - requirement. Follow the live-schema/returned-handle owner in installed + label is not native role enforcement and cannot evade a receipt requirement + while the worker gate is armed. Follow the live-schema/returned-handle owner in installed `pabcd/references/delegation.md`; never invent unsupported fields or IDs. - **EVIDENCE-TERMINAL-01 (DEFAULT, 260826).** The evidence gate blocks at most `MAX_ATTEMPTS` times per `(agent, turn)`, then RELEASES with an unresolved verdict From 63f3b0e55b43866a2499e1c9c544f7d287954af3 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 00:35:26 +0900 Subject: [PATCH 15/90] test(pabcd-state): lock explicit trigger requests (#250) --- .../components/pabcd-state/test/hook.test.ts | 270 ++++++++++-------- 1 file changed, 153 insertions(+), 117 deletions(-) diff --git a/plugins/codexclaw/components/pabcd-state/test/hook.test.ts b/plugins/codexclaw/components/pabcd-state/test/hook.test.ts index dd977045..1d398f9d 100644 --- a/plugins/codexclaw/components/pabcd-state/test/hook.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/hook.test.ts @@ -60,43 +60,123 @@ function ups(prompt: string, cwd: string, sessionId: string, turnId?: string): U }; } -test("detectTrigger: explicit triggers map to phases (EN + Korean)", () => { - assert.equal(detectTrigger("please interview me"), "I"); - assert.equal(detectTrigger("인터뷰 시작하자"), "I"); - assert.equal(detectTrigger("orchestrate I"), "I"); - assert.equal(detectTrigger("orchestrate P now"), "P"); - assert.equal(detectTrigger("plan this feature"), "P"); - assert.equal(detectTrigger("계획 세워줘"), "P"); - assert.equal(detectTrigger("orchestrate A"), "A"); - assert.equal(detectTrigger("audit this plan"), "A"); - assert.equal(detectTrigger("이거 감사해줘"), "A"); - assert.equal(detectTrigger("orchestrate B"), "B"); - assert.equal(detectTrigger("build this"), "B"); - assert.equal(detectTrigger("이거 구현해"), "B"); - assert.equal(detectTrigger("orchestrate C"), "C"); - assert.equal(detectTrigger("check this output"), "C"); - assert.equal(detectTrigger("검증 좀"), "C"); +test("detectTrigger: explicit CodexClaw phase requests map to phases", () => { + for (const [prompt, phase] of [ + ["Use cxc-pabcd to start Interview phase", "I"], + ["cxc-pabcd로 인터뷰 시작해", "I"], + ["orchestrate I", "I"], + ["Use cxc-pabcd to start Plan phase", "P"], + ["cxc-pabcd로 계획 진행해", "P"], + ["orchestrate P now", "P"], + ["Run cxc-pabcd Audit phase", "A"], + ["cxc-pabcd로 감사 진행해", "A"], + ["orchestrate A", "A"], + ["Run cxc-pabcd Build phase", "B"], + ["cxc-pabcd로 구현 진행해", "B"], + ["orchestrate B", "B"], + ["Run cxc-pabcd Check phase", "C"], + ["cxc-pabcd로 검증 진행해", "C"], + ["orchestrate C", "C"], + ] as const) assert.equal(detectTrigger(prompt), phase, prompt); +}); + +test("detectTrigger: phase priority applies only within an explicit request line", () => { + assert.equal(detectTrigger("Use cxc-pabcd to start Interview then Plan phase"), "I"); + assert.equal(detectTrigger("Summarize interview notes\nUse cxc-pabcd to start Plan phase"), "P"); +}); + +test("detectTrigger: ordinary words stay silent", () => { + for (const prompt of ["just a normal message", "", "감사합니다", "정말 감사해요 도와주셔서", + "계획을 세워줘", "이거 감사해줘", "기능 구현해줘", "검증 좀 해줘", "please interview me"]) + assert.equal(detectTrigger(prompt), null, prompt); +}); + +const ISSUE_250_PROMPTS = [ + ["order_line", "Keep going until Done means holds. ..."], + ["wn_workflow_name", "... workflow 인터뷰엔진 must keep its name."], + ["author_ko_build", "이 기능 구현해 두고 결과 보고해"], + ["author_ko_verify", "검증해 보고 알려줘"], + ["author_ko_finish", "끝까지 진행해"], + ["english_mention", "Summarize the interview notes in file X"], + ["neg_thanks", "감사합니다"], + ["neg_for_loop", "fix the for loop bug in parser.ts"], + ["neg_plain", "list the files in out/"], +] as const; + +test("issue 250: nine reported prompts stay silent", () => { + for (const [label, prompt] of ISSUE_250_PROMPTS) { + assert.equal(detectTrigger(prompt), null, label); + assert.equal(detectLoopArmRequest(prompt), false, label); + } }); -test("detectTrigger: interview wins over plan when both present", () => { - assert.equal(detectTrigger("interview then plan this"), "I"); +test("issue 250: incidental prompt emits no context and does not arm", () => { + for (const [label, prompt] of ISSUE_250_PROMPTS) { + const cwd = freshCwd(); + try { + writeState(cwd, defaultState(label)); + const state = readState(cwd, label); + assert.equal(handleUserPromptSubmit(ups(prompt, cwd, label, "t1")), "", label); + const after = readState(cwd, label); + assert.equal(after.loopArmSeen, false, label); + assert.deepEqual(after.injectedTurns, state.injectedTurns, label); + assert.deepEqual(after, state, label); + } finally { rmSync(cwd, { recursive: true, force: true }); } + } }); -test("detectTrigger: non-trigger -> null", () => { - assert.equal(detectTrigger("just a normal message"), null); - assert.equal(detectTrigger(""), null); +test("issue 250: explicit skill request arms once", () => { + const cwd = freshCwd(); + try { + const prompt = "Use [$cxc-pabcd](skill:///Users/jun/.codex/plugins/cache/codexclaw/codexclaw/0.2.39+codex.20260924082502/skills/pabcd/SKILL.md) to start Plan phase"; + assert.equal(detectTrigger(prompt), "P"); + const first = handleUserPromptSubmit(ups(prompt, cwd, "explicit-phase", "t1")); + assert.match(first, /codexclaw: (INTERVIEW|PLAN)/); + assert.equal(handleUserPromptSubmit(ups(prompt, cwd, "explicit-phase", "t1")), ""); + assert.deepEqual(readState(cwd, "explicit-phase").injectedTurns, ["t1"]); + assert.equal(detectLoopArmRequest("Run cxc-loop for this task"), true); + assert.match(handleUserPromptSubmit(ups("Run cxc-loop for this task", cwd, "explicit-loop", "t1")), /arming mandate/); + assert.equal(readState(cwd, "explicit-loop").loopArmSeen, true); + } finally { rmSync(cwd, { recursive: true, force: true }); } }); -test("detectTrigger: everyday Korean words do NOT misfire (Galileo blocker #1)", () => { - assert.equal(detectTrigger("감사합니다"), null); // "thank you" must NOT trigger AUDIT - assert.equal(detectTrigger("정말 감사해요 도와주셔서"), null); +test("issue 250: inline quoted requests are data", () => { + for (const prompt of [ + 'Summarize this quoted request: "Use cxc-pabcd to start plan phase".', + 'Summarize this quoted request: “Run cxc-loop for this task”.', + "Summarize this quoted request: 'Use cxc-pabcd to start plan phase'.", + "Summarize this quoted request: `Run cxc-loop for this task`.", + "Summarize `cxc-loop`로 written instructions.", + 'Summarize: "Run cxc-loop for this task" and "Use cxc-pabcd to start plan phase".', + '> Run cxc-loop for this task', + '- Use cxc-pabcd to start Plan phase', + '```\nRun cxc-loop for this task\nUse cxc-pabcd to start Plan phase\n```', + 'Explain how to run `cxc-loop` from the README', + ]) { + const cwd = freshCwd(); + try { + assert.equal(detectTrigger(prompt), null, prompt); + assert.equal(detectLoopArmRequest(prompt), false, prompt); + writeState(cwd, defaultState("quoted")); + const state = readState(cwd, "quoted"); + assert.equal(handleUserPromptSubmit(ups(prompt, cwd, "quoted", "t1")), "", prompt); + assert.deepEqual(readState(cwd, "quoted"), state, prompt); + } finally { rmSync(cwd, { recursive: true, force: true }); } + } }); -test("detectTrigger: natural Korean with particles/suffixes still matches", () => { - assert.equal(detectTrigger("계획을 세워줘"), "P"); - assert.equal(detectTrigger("이거 감사해줘"), "A"); - assert.equal(detectTrigger("기능 구현해줘"), "B"); - assert.equal(detectTrigger("검증 좀 해줘"), "C"); +test("issue 250: backtick command requests and mixed lines remain explicit", () => { + for (const [prompt, phase, loop] of [ + ["Run `cxc-loop` for this task", null, true], + ["Run `cxc-loop` to update docs", null, true], + ["Use `cxc-pabcd` to start Plan phase", "P", false], + ["Use `cxc-pabcd` to plan the README", "P", false], + ["Summarize interview notes\nUse cxc-pabcd to start Plan phase", "P", false], + ["orchestrate i", "I", false], + ] as const) { + assert.equal(detectTrigger(prompt), phase, prompt); + assert.equal(detectLoopArmRequest(prompt), loop, prompt); + } }); test("phase directives use resolvable skill mentions for spawn messages", () => { @@ -173,45 +253,26 @@ test("260914: hook P output carries the architect sequence; A output carries the const WP3_ORIGINAL_C2_PROMPT = "README 계약에 맞게 기존 내부 메모 생성/목록 기능을 완성해줘. 네트워크 서버나 공개 API는 아니고 src/route.mjs와 src/service.mjs의 기존 빈 구현을 채우는 작업이야. src/store.mjs와 test/notes.test.mjs는 수정하지 마. 기존 번호 문서에 결과를 기록하고 node --test test/notes.test.mjs로 실제 검증해줘. 새 의존성/추상화/파일, goal/FSM 변경, 커밋, 서브에이전트 파견은 하지 마."; -test("wp3: original Korean C2 still reaches scoped CHECK without entering C", () => { +test("wp3: ordinary Korean C2 remains silent without a CodexClaw request", () => { for (const turn of ["t1", ""] as const) { const cwd = freshCwd(); try { const session = "wp3-original-c2"; + writeState(cwd, defaultState(session)); const before = readState(cwd, session); - assert.equal(detectTrigger(WP3_ORIGINAL_C2_PROMPT), "C"); + assert.equal(detectTrigger(WP3_ORIGINAL_C2_PROMPT), null); assert.equal(detectLoopArmRequest(WP3_ORIGINAL_C2_PROMPT), false); - const payload = ups(WP3_ORIGINAL_C2_PROMPT, cwd, session, turn); - const output = handleUserPromptSubmit(payload, "linux"); - const envelope = JSON.parse(output).hookSpecificOutput; - assert.equal(envelope.hookEventName, "UserPromptSubmit"); - const ctx = envelope.additionalContext as string; - assert.match(ctx, /^\[codexclaw: CHECK\]/); - assert.match(ctx, /No-delegation means no dispatch/); - assert.match(ctx, /No-tests forbids tests, not separately authorized build\/typecheck/); - assert.match(ctx, /Independent review needs owner applicability and dispatch permission/); - assert.match(ctx, /Report unmet review; inline review is not its proof/); - assert.match(ctx, /A lexical phase hint is not execution authority/); - assert.match(ctx, /Only if a phase transition is authorized/); - assert.match(ctx, /IPABCD: IDLE \(IDLE\)/); - assert.doesNotMatch(ctx, /pass, dispatch with|retain independent review/); - const after = readState(cwd, session); - assert.equal(after.phase, before.phase); - assert.equal(after.orchestrationActive, before.orchestrationActive); - assert.equal(after.lastInjectedPhase, before.lastInjectedPhase); - assert.deepEqual(after.flags, before.flags); - assert.equal(after.loopArmSeen, before.loopArmSeen); - assert.deepEqual(after.injectedTurns, turn ? [turn] : []); + assert.equal(handleUserPromptSubmit(ups(WP3_ORIGINAL_C2_PROMPT, cwd, session, turn)), ""); + assert.deepEqual(readState(cwd, session), before); assert.equal(existsSync(join(cwd, STATE_DIR, LEDGER_FILE)), false); - if (turn) assert.equal(handleUserPromptSubmit(payload, "linux"), ""); } finally { rmSync(cwd, { recursive: true, force: true }); } } }); test("wp3: CHECK negatives retain lexical trigger and the actual persisted phase", () => { const prompts = [ - "검증해줘. 읽기 전용으로 코드만 검토해. 수정, 테스트/빌드/타입검사, goal/FSM 변경, 서브에이전트 파견 금지.", - "Check this code by reading it only; no edits, no tests, no build, no typecheck, no goals, no FSM changes, no delegation.", + "Use cxc-pabcd to start Check phase.\n검증해줘. 읽기 전용으로 코드만 검토해. 수정, 테스트/빌드/타입검사, goal/FSM 변경, 서브에이전트 파견 금지.", + "Use cxc-pabcd to start Check phase.\nCheck this code by reading it only; no edits, no tests, no build, no typecheck, no goals, no FSM changes, no delegation.", ]; for (const prompt of prompts) { for (const phase of ["IDLE", "P", "B", "C"] as const) { @@ -254,9 +315,9 @@ test("wp3: neutral C2 remains an ordinary non-trigger control", () => { test("wp3: CHECK preserves separately allowed build and read-only state inspection", () => { for (const prompt of [ - "Check this. No-tests, but npm run build is explicitly allowed. No delegation or goal/FSM mutations.", - "검증해줘. 테스트는 금지지만 빌드와 타입검사는 허용해. goal/FSM 생성과 변경은 금지하고 상태 조회는 허용해. 파견 금지.", - "Check this read-only. No-goal/no-FSM mutations; inspect get_goal and orchestrate status only. No edits, tests, build, typecheck or delegation.", + "Use cxc-pabcd to start Check phase.\nNo-tests, but npm run build is explicitly allowed. No delegation or goal/FSM mutations.", + "cxc-pabcd로 검증 진행해.\n테스트는 금지지만 빌드와 타입검사는 허용해. goal/FSM 생성과 변경은 금지하고 상태 조회는 허용해. 파견 금지.", + "Use cxc-pabcd to start Check phase read-only.\nNo-goal/no-FSM mutations; inspect get_goal and orchestrate status only. No edits, tests, build, typecheck or delegation.", ]) { const cwd = freshCwd(); try { @@ -330,8 +391,8 @@ test("handleUserPromptSubmit: idempotent within same (session,turn)", () => { const cwd = freshCwd(); try { // loose-trigger path (parser returns null for prose) — exercises turn dedup. - const first = handleUserPromptSubmit(ups("plan this", cwd, "s1", "t1")); - const second = handleUserPromptSubmit(ups("plan this", cwd, "s1", "t1")); + const first = handleUserPromptSubmit(ups("Use cxc-pabcd to start Plan phase", cwd, "s1", "t1")); + const second = handleUserPromptSubmit(ups("Use cxc-pabcd to start Plan phase", cwd, "s1", "t1")); assert.notEqual(first, ""); assert.equal(second, ""); } finally { @@ -342,8 +403,8 @@ test("handleUserPromptSubmit: idempotent within same (session,turn)", () => { test("handleUserPromptSubmit: new turn re-injects", () => { const cwd = freshCwd(); try { - const first = handleUserPromptSubmit(ups("plan this", cwd, "s1", "t1")); - const second = handleUserPromptSubmit(ups("plan this", cwd, "s1", "t2")); + const first = handleUserPromptSubmit(ups("Use cxc-pabcd to start Plan phase", cwd, "s1", "t1")); + const second = handleUserPromptSubmit(ups("Use cxc-pabcd to start Plan phase", cwd, "s1", "t2")); assert.notEqual(first, ""); assert.notEqual(second, ""); } finally { @@ -354,8 +415,8 @@ test("handleUserPromptSubmit: new turn re-injects", () => { test("handleUserPromptSubmit: different sessions are independent", () => { const cwd = freshCwd(); try { - const a = handleUserPromptSubmit(ups("plan this", cwd, "alpha", "t1")); - const b = handleUserPromptSubmit(ups("plan this", cwd, "beta", "t1")); + const a = handleUserPromptSubmit(ups("Use cxc-pabcd to start Plan phase", cwd, "alpha", "t1")); + const b = handleUserPromptSubmit(ups("Use cxc-pabcd to start Plan phase", cwd, "beta", "t1")); assert.notEqual(a, ""); assert.notEqual(b, ""); } finally { @@ -458,40 +519,19 @@ test("posix arming directive is byte-identical to its pinned snapshot", () => { assert.equal(loopArmDirective("darwin"), expected); }); -test("ORCH-MANDATE-01: detectLoopArmRequest catches loop/goalplan/continue-until-done intent (EN+KO)", () => { - assert.equal(detectLoopArmRequest("cxc-loop로 진행하자"), true); - assert.equal(detectLoopArmRequest("HOTL 모드로 돌려줘"), true); - assert.equal(detectLoopArmRequest("goalplan 잡고 시작해"), true); - assert.equal(detectLoopArmRequest("골플랜부터 등록해"), true); - assert.equal(detectLoopArmRequest("continue until done, no pauses"), true); - assert.equal(detectLoopArmRequest("루프 돌려서 처리해"), true); - assert.equal(detectLoopArmRequest("알아서 끝까지 해줘"), true); - assert.equal(detectLoopArmRequest("멈추지 말고 진행해"), true); - // Negatives: code-talk about loops must NOT arm PABCD ceremony. - assert.equal(detectLoopArmRequest("fix the for loop in parser.ts"), false); - assert.equal(detectLoopArmRequest("이 loop 버그 좀 봐줘"), false); - assert.equal(detectLoopArmRequest("루프백 오디오 설정"), false); - assert.equal(detectLoopArmRequest("계속해"), false); -}); - -test("ORCH-ARM-PABCD-01: pabcd + strong run/repeat marker arms; questions/repeat-runs do not (260714)", () => { - // Positives — natural phrasings for "run PABCD repeatedly". - assert.equal(detectLoopArmRequest("pabcd 여러 번 돌려서 해결해"), true); - assert.equal(detectLoopArmRequest("PABCD를 여러 번 돌려서 이 문제 해결해라"), true); - assert.equal(detectLoopArmRequest("run pabcd repeatedly until this is fixed"), true); - assert.equal(detectLoopArmRequest("pabcd multiple times please"), true); - assert.equal(detectLoopArmRequest("ipabcd 사이클로 돌리자"), true); - assert.equal(detectLoopArmRequest("여러 번 반복해서 해결해"), true); - // Negatives — questions ABOUT pabcd and ordinary repeat-run asks must stay cold. - assert.equal(detectLoopArmRequest("what is pabcd?"), false); - assert.equal(detectLoopArmRequest("pabcd 문서 다시 보여줘"), false); - assert.equal(detectLoopArmRequest("explain how pabcd runs internally"), false); - assert.equal(detectLoopArmRequest("pabcd가 뭐야? 계속 헷갈리네"), false); - assert.equal(detectLoopArmRequest("이 함수 여러 번 호출되는 버그 고쳐"), false); - assert.equal(detectLoopArmRequest("이 테스트 여러 번 실행해봐"), false); - assert.equal(detectLoopArmRequest("앱 아이콘 여러 번 실행해도 안 열려"), false); - assert.equal(detectLoopArmRequest("빌드 반복 실행해서 flaky 잡아줘"), false); - assert.equal(detectLoopArmRequest("여러 번 진행된 마이그레이션 롤백해줘"), false); +test("ORCH-MANDATE-01: explicit loop requests arm; incidental persistence does not", () => { + for (const prompt of ["Run cxc-loop for this task", "cxc-loop로 진행하자", "HOTL 모드로 돌려줘", + "Start goalplan for this task", "골플랜부터 등록해", "run PABCD repeatedly until this is fixed", + "PABCD를 여러 번 돌려서 이 문제 해결해라", "ipabcd 사이클로 돌리자"]) + assert.equal(detectLoopArmRequest(prompt), true, prompt); + for (const prompt of ["cxc-loop", "goalplan", "continue until done, no pauses", + "루프 돌려서 처리해", "알아서 끝까지 해줘", "멈추지 말고 진행해", "여러 번 반복해서 해결해", + "fix the for loop in parser.ts", "이 loop 버그 좀 봐줘", "루프백 오디오 설정", "계속해", + "what is pabcd?", "pabcd 문서 다시 보여줘", "explain how pabcd runs internally", + "pabcd가 뭐야? 계속 헷갈리네", "이 함수 여러 번 호출되는 버그 고쳐", + "이 테스트 여러 번 실행해봐", "앱 아이콘 여러 번 실행해도 안 열려", + "빌드 반복 실행해서 flaky 잡아줘", "여러 번 진행된 마이그레이션 롤백해줘"]) + assert.equal(detectLoopArmRequest(prompt), false, prompt); }); test("260714 wp3: loop-arm prompt persists loopArmSeen on the un-armed branch (even turnless)", () => { @@ -514,7 +554,7 @@ test("260714 wp3: loop-arm prompt persists loopArmSeen on the un-armed branch (e test("040: trigger + loop phrase on an un-armed FSM yields the mandate, not a phase", () => { const cwd = freshCwd(); try { - const out = handleUserPromptSubmit(ups("plan this and then 루프 돌려서 끝까지 해줘", cwd, "la3", "t1")); + const out = handleUserPromptSubmit(ups("Use cxc-pabcd to start Plan phase and run cxc-loop for this task", cwd, "la3", "t1")); const ctx = JSON.parse(out.trimEnd()).hookSpecificOutput.additionalContext as string; assert.match(ctx, /arming mandate/); const st = readState(cwd, "la3"); @@ -563,7 +603,7 @@ test("ORCH-MANDATE-01: loop request against un-armed FSM injects the arming mand test("040: on an un-armed FSM the loop-arm mandate wins over a phase trigger", () => { const cwd = freshCwd(); try { - const out = handleUserPromptSubmit(ups("plan this and then loop until done", cwd, "s1", "t1")); + const out = handleUserPromptSubmit(ups("Use cxc-pabcd to start Plan phase and run cxc-loop until done", cwd, "s1", "t1")); const parsed = JSON.parse(out.trimEnd()); const ctx = parsed.hookSpecificOutput.additionalContext as string; assert.match(ctx, /arming mandate/); @@ -591,12 +631,10 @@ test("ORCH-MANDATE-01: loop-arm and agbrowse directives compose when both are re test("wp3: loop arming output is scope-first and does not activate a phase", () => { for (const prompt of [ - "cxc-loop", - "cxc-loop, interview-only; do not create a goal", - "cxc-loop, plan-only; no implementation", + "Run cxc-loop, interview-only; do not create a goal", + "Run cxc-loop, plan-only; no implementation", "cxc-loop로 인터뷰만 해줘. goal 만들지 마", "cxc-loop로 계획만 작성해줘. 구현하지 마", - "Explain the quoted example cxc-loop; read-only, no FSM changes", ]) { const cwd = freshCwd(); try { @@ -619,11 +657,9 @@ test("wp3: loop arming output is scope-first and does not activate a phase", () test("wp3: arming limits precede recipes on both platforms and never arm a phase", () => { for (const platform of ["linux", "win32"] as const) { for (const prompt of [ - "cxc-loop", - "cxc-loop, plan-only; no-goal, no-FSM, no-tests, no-delegation; read-only", + "Run cxc-loop, plan-only; no-goal, no-FSM, no-tests, no-delegation; read-only", "cxc-loop로 인터뷰만 해줘. goal/FSM 변경, 테스트, 수정, 파견 금지.", - "Explain the quoted cxc-loop example; read-only, no-goal, no-FSM, no-tests, no-delegation.", - "cxc-loop, explicit HITL; no-delegation; tests only on macmini, no local tests", + "Run cxc-loop, explicit HITL; no-delegation; tests only on macmini, no local tests", ]) { const cwd = freshCwd(); try { @@ -674,7 +710,7 @@ test("handleUserPromptSubmit: agbrowse request is idempotent within same turn", test("handleUserPromptSubmit: PABCD hint wins over agbrowse without phase entry", () => { const cwd = freshCwd(); try { - const out = handleUserPromptSubmit(ups("plan this with agbrowse", cwd, "s1", "t1")); + const out = handleUserPromptSubmit(ups("Use cxc-pabcd to start Plan phase with agbrowse", cwd, "s1", "t1")); const ctx = JSON.parse(out).hookSpecificOutput.additionalContext as string; assert.equal(ctx, withFooter(`${interviewDirective()}\n\n${TRIGGER_AUTHORITY_NOTE}`, "IDLE")); assert.doesNotMatch(ctx, /agbrowse fetch/); @@ -691,7 +727,7 @@ test("wp4: interview policy off selects PLAN advice without phase entry", () => const cwd = freshCwd(); try { writeFileSync(join(cwd, "codexclaw.json"), JSON.stringify({ interview: "off" }), "utf8"); - const out = handleUserPromptSubmit(ups("plan this with agbrowse", cwd, "s1off", "t1")); + const out = handleUserPromptSubmit(ups("Use cxc-pabcd to start Plan phase with agbrowse", cwd, "s1off", "t1")); const ctx = JSON.parse(out).hookSpecificOutput.additionalContext as string; assert.equal(ctx, withFooter(`${phaseDirective("P")}\n\n${TRIGGER_AUTHORITY_NOTE}`, "IDLE")); assert.doesNotMatch(ctx, /agbrowse fetch/); @@ -842,7 +878,7 @@ test("L3b: same-turn re-fire does NOT double-append the ledger", () => { test("L3b: no command falls through to advisory detectTrigger without a transition", () => { const cwd = freshCwd(); try { - const out = handleUserPromptSubmit(ups("plan this feature", cwd, "s7", "t1")); + const out = handleUserPromptSubmit(ups("Use cxc-pabcd to start Plan phase for this feature", cwd, "s7", "t1")); assert.equal(JSON.parse(out).hookSpecificOutput.additionalContext, withFooter(`${interviewDirective()}\n\n${TRIGGER_AUTHORITY_NOTE}`, "IDLE")); const state = readState(cwd, "s7"); @@ -973,7 +1009,7 @@ test("chat D-close succeeds once the tasks are done", () => { test("040: a natural-language build trigger from IDLE leaves the phase alone", () => { const cwd = freshCwd(); try { - const out = handleUserPromptSubmit(ups("이거 구현해줘", cwd, "ta1", "t1")); + const out = handleUserPromptSubmit(ups("cxc-pabcd로 구현 진행해줘", cwd, "ta1", "t1")); const ctx = JSON.parse(out.trimEnd()).hookSpecificOutput.additionalContext as string; assert.match(ctx, /BUILD/); assert.match(ctx, /TRIGGER-AUTHORITY-01/); @@ -987,9 +1023,9 @@ test("040: a natural-language build trigger from IDLE leaves the phase alone", ( test("wp3: plain P/I hints never enter or advance, including explicit no-FSM", () => { for (const prompt of [ - "plan this", "interview me", "계획을 세워줘", "인터뷰만 해줘", - "Plan this read-only; no FSM mutations, goals, tests or delegation.", - "인터뷰만 해줘. FSM 변경, goal 생성, 파일 수정, 테스트, 서브에이전트 파견 금지.", + "Use cxc-pabcd to start Plan phase", "Use cxc-pabcd to start Interview phase", "cxc-pabcd로 계획 진행해", "cxc-pabcd로 인터뷰 시작해", + "Use cxc-pabcd to start Plan phase read-only; no FSM mutations, goals, tests or delegation.", + "cxc-pabcd로 인터뷰 시작해. FSM 변경, goal 생성, 파일 수정, 테스트, 서브에이전트 파견 금지.", ]) { for (const phase of ["IDLE", "P", "B"] as const) { for (const turn of ["t1", ""] as const) { @@ -1025,7 +1061,7 @@ test("040: a mid-cycle trigger cannot move the phase and the footer reports the const cwd = freshCwd(); try { writeState(cwd, { ...defaultState("ta3"), phase: "P", orchestrationActive: true, lastInjectedPhase: "P" }); - const out = handleUserPromptSubmit(ups("이거 구현해줘", cwd, "ta3", "t1")); + const out = handleUserPromptSubmit(ups("cxc-pabcd로 구현 진행해줘", cwd, "ta3", "t1")); const ctx = JSON.parse(out.trimEnd()).hookSpecificOutput.additionalContext as string; assert.match(ctx, /TRIGGER-AUTHORITY-01/); assert.match(ctx, /IPABCD: P/); // the phase on disk, not the one asked for @@ -1075,7 +1111,7 @@ test("040: same through mode 3 (same phase, header only)", () => { test("040: turnless hints preserve phase while loop requests persist bookkeeping", () => { const cwdA = freshCwd(); try { - const out = handleUserPromptSubmit(ups("plan this", cwdA, "tl1", "")); + const out = handleUserPromptSubmit(ups("Use cxc-pabcd to start Plan phase", cwdA, "tl1", "")); assert.match(JSON.parse(out).hookSpecificOutput.additionalContext, /IPABCD: IDLE \(IDLE\)/); const state = readState(cwdA, "tl1"); assert.equal(state.phase, "IDLE"); From 8a2609af6aa067a888ef7c6a0d8c656c6c2e2c89 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 00:35:44 +0900 Subject: [PATCH 16/90] fix(pabcd-state): require explicit phase and loop requests (#250) --- .../components/pabcd-state/src/hook.ts | 86 +++++++++++-------- 1 file changed, 49 insertions(+), 37 deletions(-) diff --git a/plugins/codexclaw/components/pabcd-state/src/hook.ts b/plugins/codexclaw/components/pabcd-state/src/hook.ts index 537df632..5974ffff 100644 --- a/plugins/codexclaw/components/pabcd-state/src/hook.ts +++ b/plugins/codexclaw/components/pabcd-state/src/hook.ts @@ -230,22 +230,50 @@ export function resolveCxcInDirective(text: string): string { } } -/** - * Detect an explicit IPABCD/interview trigger. Explicit only — no goal-mode - * branch (A3 decision, see 022.3). Both English and Korean phrasings. - * Order matters: interview is checked first so "orchestrate i" wins over "p". - */ +/** Lines that can carry an advisory CodexClaw request, excluding quoted examples. */ +function requestLines(prompt: string): string[] { + const result: string[] = []; + let fenced = false; + for (const raw of (prompt ?? "").split(/\r?\n/)) { + const line = raw.trim(); + if (/^```/.test(line)) { fenced = !fenced; continue; } + if (fenced || !line || /^(?:>|[-*] |\d+[.)] )/.test(line)) continue; + // Explanatory leads describe a command; later mentions of docs/README do not. + const explanatory = /^(?:(?:please|좀)\s+)?(?:explain|describe|how do|how to|what is|what does|why)\b|^(?:좀\s*)?(?:설명|어떻게|뭐야)/i.test(line); + const unquoted = line + .replace(/`([^`]*)`/g, (match, inner: string, offset: number) => { + if (explanatory || !/^(?:\$?(?:codexclaw:)?cxc-(?:loop|pabcd)|orchestrate\s+[ipabc])$/i.test(inner.trim())) return " "; + const before = line.slice(0, offset); + const after = line.slice(offset + match.length); + const addressed = /^(?:(?:please|좀)\s+)?(?:run|use|start|invoke|실행|돌려)\s*$/i.test(before) + || (/^(?:(?:please|좀)\s*)?$/i.test(before) && /^\s*(?:로|으로|써서)/.test(after)); + return addressed ? inner : " "; + }) + .replace(/"(?:\\.|[^"\\])*"|“[^”]*”|(? Date: Mon, 28 Sep 2026 00:37:27 +0900 Subject: [PATCH 17/90] fix(pabcd-state): persist per-turn Stop cap fields (#254) --- .../components/pabcd-state/src/state.ts | 15 +++++++++-- .../components/pabcd-state/test/state.test.ts | 27 +++++++++++++++++++ 2 files changed, 40 insertions(+), 2 deletions(-) diff --git a/plugins/codexclaw/components/pabcd-state/src/state.ts b/plugins/codexclaw/components/pabcd-state/src/state.ts index 066db540..004c6271 100644 --- a/plugins/codexclaw/components/pabcd-state/src/state.ts +++ b/plugins/codexclaw/components/pabcd-state/src/state.ts @@ -133,8 +133,12 @@ export interface State { * but only a cursor answers "is this new". */ stopMetricCursor: number; - /** Stop blocks this session, never reset — the loop's absolute bound. */ + /** Stop blocks in the current observed user turn. */ stopBlockTotal: number; + /** Last genuine UserPromptSubmit turn whose total was reset. */ + stopBlockTurnId: string | null; + /** Whether the current turn's absolute-cap notice was emitted. */ + stopBlockCapNotified: boolean; // 260714 wp3 (IDLE-EDIT-ADVISORY-01): true once this session saw a loop-arm request // (detectLoopArmRequest). Retained across D-close (multi-cycle re-arm nudge is the // feature); cleared only by explicit reset (operator stand-down). @@ -298,6 +302,8 @@ export function defaultState(sessionId: string, slug = ""): State { stopBlockWorkPhaseId: null, stopMetricCursor: 0, stopBlockTotal: 0, + stopBlockTurnId: null, + stopBlockCapNotified: false, loopArmSeen: false, idleEditNudges: 0, memoryWriteRequested: false, @@ -317,7 +323,7 @@ function sessionsDir(cwd: string): string { return join(cwd, STATE_DIR, SESSIONS_SUBDIR); } -function statePath(cwd: string, sessionId: string): string { +export function statePath(cwd: string, sessionId: string): string { return join(sessionsDir(cwd), `${sanitizeKey(sessionId)}.json`); } @@ -552,6 +558,11 @@ export function readStateStrict(cwd: string, sessionId: string): { state: State; typeof parsed.stopBlockTotal === "number" && Number.isFinite(parsed.stopBlockTotal) && parsed.stopBlockTotal >= 0 ? Math.floor(parsed.stopBlockTotal) : 0, + stopBlockTurnId: + typeof parsed.stopBlockTurnId === "string" && parsed.stopBlockTurnId.length > 0 + ? parsed.stopBlockTurnId + : null, + stopBlockCapNotified: parsed.stopBlockCapNotified === true, // 260714 wp3: strict reconstruction (old files read false/0 — backward-compatible). loopArmSeen: parsed.loopArmSeen === true, idleEditNudges: diff --git a/plugins/codexclaw/components/pabcd-state/test/state.test.ts b/plugins/codexclaw/components/pabcd-state/test/state.test.ts index 405ab7c9..40668fd7 100644 --- a/plugins/codexclaw/components/pabcd-state/test/state.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/state.test.ts @@ -47,6 +47,8 @@ test("SessionStart ensureState: fresh session creates the exact default IDLE sta stopBlockWorkPhaseId: null, stopMetricCursor: 0, stopBlockTotal: 0, + stopBlockTurnId: null, + stopBlockCapNotified: false, loopArmSeen: false, idleEditNudges: 0, memoryWriteRequested: false, @@ -67,6 +69,31 @@ test("SessionStart ensureState: fresh session creates the exact default IDLE sta } }); +test("stopBlockTurnId round trips and malformed value becomes null", () => { + const cwd = freshCwd(); + try { + writeState(cwd, { ...defaultState("turn-state"), stopBlockTurnId: "turn-new" }); + assert.equal(readState(cwd, "turn-state").stopBlockTurnId, "turn-new"); + const file = join(cwd, STATE_DIR, SESSIONS_SUBDIR, "turn-state.json"); + const persisted = JSON.parse(readFileSync(file, "utf8")); + writeFileSync(file, JSON.stringify({ ...persisted, stopBlockTurnId: 42 })); + assert.equal(readState(cwd, "turn-state").stopBlockTurnId, null); + } finally { rmSync(cwd, { recursive: true, force: true }); } +}); + +test("stopBlockCapNotified defaults to false and round-trips", () => { + const cwd = freshCwd(); + try { + assert.equal(defaultState("notice").stopBlockCapNotified, false); + writeState(cwd, { ...defaultState("notice"), stopBlockCapNotified: true }); + assert.equal(readState(cwd, "notice").stopBlockCapNotified, true); + const file = join(cwd, STATE_DIR, SESSIONS_SUBDIR, "notice.json"); + const persisted = JSON.parse(readFileSync(file, "utf8")); + writeFileSync(file, JSON.stringify({ ...persisted, stopBlockCapNotified: "true" })); + assert.equal(readState(cwd, "notice").stopBlockCapNotified, false); + } finally { rmSync(cwd, { recursive: true, force: true }); } +}); + test("050: a corrupt snapshot is rejected rather than coerced", () => { const cwd = freshCwd(); try { From 880ba6eaf613c84db1fb7b315e3c2a0d285a5b0b Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 00:37:37 +0900 Subject: [PATCH 18/90] fix(pabcd-state): reset Stop cap per user turn and notify once (#254) --- .../components/pabcd-state/src/hook.ts | 26 +++-- .../test/hook-continuation.test.ts | 102 +++++++++++++++++- plugins/codexclaw/test/hook-e2e.test.mjs | 2 + 3 files changed, 120 insertions(+), 10 deletions(-) diff --git a/plugins/codexclaw/components/pabcd-state/src/hook.ts b/plugins/codexclaw/components/pabcd-state/src/hook.ts index 537df632..3ba2bc3d 100644 --- a/plugins/codexclaw/components/pabcd-state/src/hook.ts +++ b/plugins/codexclaw/components/pabcd-state/src/hook.ts @@ -26,6 +26,7 @@ import { matchesDcloseRecovery, readState, STATE_DIR, + statePath, writeState, type Phase, type State, @@ -649,7 +650,11 @@ export function handleUserPromptSubmit( // `cxc memory allow-write`; it must never break prompt handling. } } - const state = readState(payload.cwd, payload.session_id); + let state = readState(payload.cwd, payload.session_id); + if (turn && existsSync(statePath(payload.cwd, payload.session_id)) && state.stopBlockTurnId !== turn) { + state = { ...state, stopBlockTotal: 0, stopBlockTurnId: turn, stopBlockCapNotified: false }; + writeState(payload.cwd, state); + } if (turn && state.injectedTurns.includes(turn)) return ""; // L3b: parser-first AUTHORITATIVE path. An explicit, line-anchored @@ -1456,7 +1461,7 @@ function observeProgress(cwd: string, state: State): ProgressObservation { return { progressed, metricCursor, workPhaseId }; } -function bumpStopCounter(cwd: string, state: State): number | "release" { +function bumpStopCounter(cwd: string, state: State): number | "phase-cap" | "total-cap" | "total-cap-silent" { const obs = observeProgress(cwd, state); const nextCount = obs.progressed ? 1 : state.stopBlockCount + 1; const nextTotal = state.stopBlockTotal + 1; @@ -1465,9 +1470,12 @@ function bumpStopCounter(cwd: string, state: State): number | "release" { // regardless of how often progress recharges the per-phase counter. const carry = { stopMetricCursor: obs.metricCursor, stopBlockTotal: nextTotal }; if (nextCount > MAX_STOP_BLOCKS || nextTotal > MAX_STOP_BLOCKS_TOTAL) { - // give up the loop: reset the counter and release so the turn can end. - writeState(cwd, { ...state, ...carry, stopBlockPhase: null, stopBlockWorkPhaseId: null, stopBlockCount: 0 }); - return "release"; + const totalCap = nextTotal > MAX_STOP_BLOCKS_TOTAL; + const alreadyNotified = state.stopBlockCapNotified === true; + writeState(cwd, { ...state, ...carry, stopBlockPhase: null, stopBlockWorkPhaseId: null, + stopBlockCount: 0, stopBlockCapNotified: totalCap ? true : state.stopBlockCapNotified }); + if (!totalCap) return "phase-cap"; + return alreadyNotified ? "total-cap-silent" : "total-cap"; } writeState(cwd, { ...state, @@ -1788,7 +1796,9 @@ export function handleStop( if (!goalActive) return ""; // bail: don't pile on during context-pressure/compaction recovery. if (isContextPressureTail(readTranscriptTail(payload.transcript_path))) return ""; - if (bumpStopCounter(payload.cwd, state) === "release") return ""; + const count = bumpStopCounter(payload.cwd, state); + if (count === "total-cap") return `${JSON.stringify({ systemMessage: "CodexClaw Stop continuation cap (24) reached for this user turn; releasing." })}\n`; + if (typeof count !== "number") return ""; return buildGoalIdleBlock(payload.cwd, state, payload.session_id, platform); } @@ -1807,7 +1817,9 @@ export function handleStop( // bail: don't pile on during context-pressure/compaction recovery. if (isContextPressureTail(readTranscriptTail(payload.transcript_path))) return ""; - if (bumpStopCounter(payload.cwd, state) === "release") return ""; + const count = bumpStopCounter(payload.cwd, state); + if (count === "total-cap") return `${JSON.stringify({ systemMessage: "CodexClaw Stop continuation cap (24) reached for this user turn; releasing." })}\n`; + if (typeof count !== "number") return ""; const plateau = objectivePlateau(payload.cwd, payload.session_id); if (plateau.flat) return buildPlateauDivergeBlock(state.phase, plateau, payload.cwd, payload.session_id); // 040: enrich the block reason with goalplan-derived remaining work (text-only, after diff --git a/plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts b/plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts index 341cff01..859901e6 100644 --- a/plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts @@ -1040,12 +1040,16 @@ test("050 S10/S14: the absolute cap holds against forged progress", () => { try { withGoalsDb([{ thread_id: "s10", status: "active" }], () => { metricSession(cwd, "s10"); - let released = 0; + let notice = ""; for (let i = 1; i <= MAX_STOP_BLOCKS_TOTAL + 1; i++) { record(cwd, "s10", "score", i); - if (handleStop(stop(cwd, "s10")) === "") released = i; + const out = handleStop(stop(cwd, "s10")); + if (i <= MAX_STOP_BLOCKS_TOTAL) assert.equal(JSON.parse(out).decision, "block", `block ${i}`); + else notice = out; } - assert.equal(released, MAX_STOP_BLOCKS_TOTAL + 1, "released exactly at the absolute cap"); + const parsed = JSON.parse(notice); + assert.match(parsed.systemMessage, /continuation cap \(24\) reached/); + assert.equal(parsed.decision, undefined, "the notice does not block"); const st = readState(cwd, "s10"); assert.equal(st.stopBlockTotal, MAX_STOP_BLOCKS_TOTAL + 1, "the total never resets"); assert.equal(st.stopMetricCursor, MAX_STOP_BLOCKS_TOTAL + 1, "the cursor advances on release too"); @@ -1053,6 +1057,98 @@ test("050 S10/S14: the absolute cap holds against forged progress", () => { } finally { rmSync(cwd, { recursive: true, force: true }); } }); +test("absolute Stop cap resets once on new real UserPromptSubmit turn", () => { + const cwd = freshCwd(); + try { + withGoalsDb([{ thread_id: "turn-reset", status: "active" }], () => { + writeState(cwd, { ...defaultState("turn-reset"), phase: "B", orchestrationActive: true, + stopBlockTotal: 24, stopBlockTurnId: "old", stopBlockCapNotified: true, + stopBlockCount: 2, stopMetricCursor: 7 }); + handleUserPromptSubmit(ups("continue", cwd, "turn-reset", "new")); + let state = readState(cwd, "turn-reset"); + assert.equal(state.stopBlockTotal, 0); + assert.equal(state.stopBlockTurnId, "new"); + assert.equal(state.stopBlockCapNotified, false); + assert.equal(state.stopBlockCount, 2, "a new turn preserves phase progress"); + assert.equal(state.stopMetricCursor, 7, "a new turn preserves observed metrics"); + assert.equal(JSON.parse(handleStop(stop(cwd, "turn-reset"))).decision, "block"); + handleUserPromptSubmit(ups("continue", cwd, "turn-reset", "new")); + assert.equal(readState(cwd, "turn-reset").stopBlockTotal, 1, "duplicate prompt cannot reset"); + handleUserPromptSubmit(ups("continue", cwd, "turn-reset", "next")); + state = readState(cwd, "turn-reset"); + assert.equal(state.stopBlockTotal, 0); + assert.equal(state.stopBlockTurnId, "next"); + }); + } finally { rmSync(cwd, { recursive: true, force: true }); } +}); + +test("UserPromptSubmit without SessionStart does not create a state file for the cap reset", () => { + const cwd = freshCwd(); + try { + handleUserPromptSubmit(ups("continue", cwd, "missing-state", "new")); + assert.equal(readState(cwd, "missing-state").stopBlockTurnId, null); + assert.equal(readState(cwd, "missing-state").stopBlockTotal, 0); + } finally { rmSync(cwd, { recursive: true, force: true }); } +}); + +test("Stop continuation never invokes reset path", () => { + const cwd = freshCwd(); + try { + withGoalsDb([{ thread_id: "same-turn", status: "active" }], () => { + metricSession(cwd, "same-turn"); + handleUserPromptSubmit(ups("continue", cwd, "same-turn", "t1")); + for (let i = 1; i <= MAX_STOP_BLOCKS_TOTAL; i++) { + record(cwd, "same-turn", "score", i); + assert.equal(JSON.parse(handleStop(stop(cwd, "same-turn", true))).decision, "block"); + } + record(cwd, "same-turn", "score", 25); + const out = JSON.parse(handleStop(stop(cwd, "same-turn", true))); + assert.match(out.systemMessage, /continuation cap \(24\) reached/); + assert.equal(out.decision, undefined); + assert.equal(readState(cwd, "same-turn").stopBlockTotal, 25); + }); + } finally { rmSync(cwd, { recursive: true, force: true }); } +}); + +for (const turnId of ["t1", undefined]) { + test(`cap notice once per turn (${turnId ? "turn id" : "no turn id"})`, () => { + const cwd = freshCwd(); + try { + withGoalsDb([{ thread_id: "notice-once", status: "active" }], () => { + metricSession(cwd, "notice-once"); + if (turnId) handleUserPromptSubmit(ups("continue", cwd, "notice-once", turnId)); + for (let i = 1; i <= MAX_STOP_BLOCKS_TOTAL + 1; i++) { + record(cwd, "notice-once", "score", i); + const payload = { ...stop(cwd, "notice-once"), turn_id: turnId }; + const out = handleStop(payload); + if (i === MAX_STOP_BLOCKS_TOTAL + 1) assert.match(JSON.parse(out).systemMessage, /continuation cap \(24\) reached/); + } + record(cwd, "notice-once", "score", 26); + assert.equal(handleStop({ ...stop(cwd, "notice-once"), turn_id: turnId }), ""); + assert.equal(readState(cwd, "notice-once").stopBlockCapNotified, true); + }); + } finally { rmSync(cwd, { recursive: true, force: true }); } + }); +} + +test("new turn re-arms the cap notice", () => { + const cwd = freshCwd(); + try { + withGoalsDb([{ thread_id: "rearm", status: "active" }], () => { + metricSession(cwd, "rearm"); + writeState(cwd, { ...readState(cwd, "rearm"), stopBlockTotal: 24, + stopBlockTurnId: "old", stopBlockCapNotified: true }); + handleUserPromptSubmit(ups("continue", cwd, "rearm", "new")); + assert.equal(readState(cwd, "rearm").stopBlockCapNotified, false); + for (let i = 1; i <= MAX_STOP_BLOCKS_TOTAL + 1; i++) { + record(cwd, "rearm", "score", i); + const out = handleStop(stop(cwd, "rearm")); + if (i === MAX_STOP_BLOCKS_TOTAL + 1) assert.match(JSON.parse(out).systemMessage, /continuation cap \(24\) reached/); + } + }); + } finally { rmSync(cwd, { recursive: true, force: true }); } +}); + test("050 S15: a truncated ledger cannot replay old observations", () => { const cwd = freshCwd(); try { diff --git a/plugins/codexclaw/test/hook-e2e.test.mjs b/plugins/codexclaw/test/hook-e2e.test.mjs index ce3c68e0..790b9c6b 100644 --- a/plugins/codexclaw/test/hook-e2e.test.mjs +++ b/plugins/codexclaw/test/hook-e2e.test.mjs @@ -186,6 +186,8 @@ test("SessionStart state bootstrap: fresh compiled hook creates exact IDLE state stopBlockWorkPhaseId: null, stopMetricCursor: 0, stopBlockTotal: 0, + stopBlockTurnId: null, + stopBlockCapNotified: false, loopArmSeen: false, idleEditNudges: 0, memoryWriteRequested: false, From d8ea3fda9016813719ca84e2fa44d09862915497 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 00:44:26 +0900 Subject: [PATCH 19/90] test: align trigger expectations with explicit requests (#250) --- .../pabcd-state/test/hook-continuation.test.ts | 8 ++++---- plugins/codexclaw/test/build.test.mjs | 4 ++-- plugins/codexclaw/test/hook-e2e.test.mjs | 10 +++++----- 3 files changed, 11 insertions(+), 11 deletions(-) diff --git a/plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts b/plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts index 341cff01..2f3ae87c 100644 --- a/plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts @@ -72,7 +72,7 @@ test("L11: active goal suppresses I-trigger (no directive, no interview state)", const cwd = freshCwd(); try { withGoalsDb([{ thread_id: "sg1", status: "active" }], () => { - const out = handleUserPromptSubmit(ups("please interview me", cwd, "sg1", "t1")); + const out = handleUserPromptSubmit(ups("Use cxc-pabcd to start Interview phase", cwd, "sg1", "t1")); assert.equal(out, "", "I-trigger must be suppressed while the native goal is active"); const st = readState(cwd, "sg1"); assert.equal(st.orchestrationActive, false, "suppressed I must not activate orchestration"); @@ -86,7 +86,7 @@ test("L11: inactive goal allows I advice without automatic phase entry", () => { const cwd = freshCwd(); try { withGoalsDb([{ thread_id: "sg2", status: "complete" }], () => { - const out = handleUserPromptSubmit(ups("please interview me", cwd, "sg2", "t1")); + const out = handleUserPromptSubmit(ups("Use cxc-pabcd to start Interview phase", cwd, "sg2", "t1")); assert.notEqual(out, "", "inactive goal must allow the interview directive"); const st = readState(cwd, "sg2"); assert.equal(st.phase, "IDLE"); @@ -204,7 +204,7 @@ test("WP4 delivery: the explicit I trigger carries the grounding rules", () => { const cwd = freshCwd(); try { writeState(cwd, { ...defaultState("gr2"), phase: "IDLE" }); - const ctx = groundingContext(handleUserPromptSubmit(ups("interview me about this", cwd, "gr2", "t-gr2"))); + const ctx = groundingContext(handleUserPromptSubmit(ups("Use cxc-pabcd to start Interview phase", cwd, "gr2", "t-gr2"))); assert.match(ctx, /INTERVIEW-GROUND-01/); assert.match(ctx, /--map/); assert.equal(readState(cwd, "gr2").phase, "IDLE"); @@ -258,7 +258,7 @@ test("wp3: I preserves Mind delivery and explicitly scopes it under no-delegatio try { withGoalsDb([], () => { const ctx = groundingContext(handleUserPromptSubmit(ups( - "Interview me only; no delegation, no tests, no implementation.", cwd, "wp3-i", "i1"))); + "Use cxc-pabcd to start Interview phase only; no delegation, no tests, no implementation.", cwd, "wp3-i", "i1"))); assert.match(ctx, /No-delegation means no dispatch/); assert.match(ctx, /This also scopes the Mind instructions below/); assert.match(ctx, /Mind dispatch/); diff --git a/plugins/codexclaw/test/build.test.mjs b/plugins/codexclaw/test/build.test.mjs index 59520497..30c9e346 100644 --- a/plugins/codexclaw/test/build.test.mjs +++ b/plugins/codexclaw/test/build.test.mjs @@ -84,14 +84,14 @@ test("compiled dist rewrote .ts import specifiers to .js", () => { } }); -test("compiled pabcd-state natural I hint emits advice and dedup without phase entry", () => { +test("compiled pabcd-state explicit I request emits advice and dedup without phase entry", () => { runBuild(); const cli = join(pluginRoot, "components", "pabcd-state", "dist", "cli.js"); const tmp = mkdtempSync(join(tmpdir(), "ccx-build-")); const home = mkdtempSync(join(tmpdir(), "ccx-build-goals-")); try { const payload = JSON.stringify({ - hook_event_name: "UserPromptSubmit", prompt: "interview me about this feature", + hook_event_name: "UserPromptSubmit", prompt: "Use cxc-pabcd to start Interview phase for this feature", cwd: tmp, session_id: "s-build-test", turn_id: "t1", }); const res = spawnSync("node", [cli, "hook", "user-prompt-submit"], { diff --git a/plugins/codexclaw/test/hook-e2e.test.mjs b/plugins/codexclaw/test/hook-e2e.test.mjs index ce3c68e0..12c11e22 100644 --- a/plugins/codexclaw/test/hook-e2e.test.mjs +++ b/plugins/codexclaw/test/hook-e2e.test.mjs @@ -572,9 +572,9 @@ test("WP7/G19: session-start provider hook e2e - exit 0 + parseable SessionStart } finally { rmSync(emptyPath, { recursive: true, force: true }); } }); -// Real registered dist entry: natural hints emit guidance/dedup only; explicit +// Real registered dist entry: explicit phase requests emit guidance/dedup only; // commands below prove legal entry. Use an isolated inactive-goal environment. -test("WP22/G19: natural plan hint emits PLAN advice with IDLE footer, never activates", () => { +test("WP22/G19: explicit plan request emits PLAN advice with IDLE footer, never activates", () => { const { event, hookEvent, distAbs } = readHookCommand("./hooks/user-prompt-submit-checking-pabcd-trigger.json"); assert.equal(event, "UserPromptSubmit"); const ep = snapshotEntrypoint(distAbs); @@ -585,7 +585,7 @@ test("WP22/G19: natural plan hint emits PLAN advice with IDLE footer, never acti writeFileSync(join(tmp, "codexclaw.json"), JSON.stringify({ interview: "off" })); const res = runHook(ep, hookEvent, { hook_event_name: "UserPromptSubmit", session_id: "s1", cwd: tmp, turn_id: "t1", - prompt: "plan this", + prompt: "Use cxc-pabcd to start Plan phase", }, home.env); assert.equal(res.status, 0, res.stderr); const out = JSON.parse(res.stdout); @@ -623,7 +623,7 @@ test("wp3: advisory snapshot then agent CLI entry reports real state and preserv const boot = runHook(ep, start.hookEvent, { hook_event_name: "SessionStart", session_id: sessionId, cwd }, home.env); assert.equal(boot.status, 0, boot.stderr); const hint = runHook(ep, prompt.hookEvent, { hook_event_name: "UserPromptSubmit", - session_id: sessionId, cwd, turn_id: "hint", prompt: phase === "P" ? "plan this" : "인터뷰만 해줘" }, home.env); + session_id: sessionId, cwd, turn_id: "hint", prompt: phase === "P" ? "Use cxc-pabcd to start Plan phase" : "Use cxc-pabcd to start Interview phase" }, home.env); assert.equal(hint.status, 0, hint.stderr); assert.match(JSON.parse(hint.stdout).hookSpecificOutput.additionalContext, /IPABCD: IDLE \(IDLE\)/); assert.match(cli("status").stdout, /phase=IDLE/); @@ -1095,7 +1095,7 @@ test("subagent-guard: user-prompt-submit with agent fields is silent and writes try { const res = runHook(ep, hookEvent, { hook_event_name: "UserPromptSubmit", session_id: "s-parent", cwd: tmp, turn_id: "t1", - prompt: "interview me, then plan this", // root hints emit guidance/dedup; child guard must remain silent + prompt: "Use cxc-pabcd to start Interview phase", // root request emits guidance/dedup; child guard must remain silent agent_id: "agent-1", agent_type: "worker", }); assert.equal(res.status, 0, res.stderr); From 82b394cbda787faccdf24a5a14616cc9d96eba37 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 00:45:20 +0900 Subject: [PATCH 20/90] refactor(pabcd-state): use the shared PABCD policy reader in the SubagentStop gate --- .../pabcd-state/src/subagent-evidence.ts | 19 +------------------ 1 file changed, 1 insertion(+), 18 deletions(-) diff --git a/plugins/codexclaw/components/pabcd-state/src/subagent-evidence.ts b/plugins/codexclaw/components/pabcd-state/src/subagent-evidence.ts index 2eb75838..1602f1e3 100644 --- a/plugins/codexclaw/components/pabcd-state/src/subagent-evidence.ts +++ b/plugins/codexclaw/components/pabcd-state/src/subagent-evidence.ts @@ -28,6 +28,7 @@ * (transcript_path = parent, agent_transcript_path = child): schema.rs:576, hook_runtime.rs:302 * - decision:"block" + reason re-prompts the child's own turn: stop.rs:263,351 + turn.rs:323 */ +import { readPabcdEnabled } from "./interview-policy.ts"; import { existsSync, lstatSync, @@ -53,7 +54,6 @@ import { type UnverifiedSubagent, } from "./state.ts"; import type { SubagentStopPayload } from "./hook.ts"; -import { configPath } from "./interview-policy.ts"; /** * agent_type values routed to this gate. @@ -469,23 +469,6 @@ export function escalationDirective(): string { * The SubagentStop decision. Returns the codex hook stdout (a `{decision:"block",reason}` * JSON string to force the child to continue, or `""` to release). Total: never throws. */ -function readPabcdEnabled(cwd: string): boolean { - // WP2-012 owns the shared reader. Keep this isolated branch buildable until its - // predecessor lands; the policy precedence and shape match that reader. - const override = process.env.CODEXCLAW_PABCD?.trim().toLowerCase(); - if (override === "off" || override === "0" || override === "false") return false; - if (override === "on" || override === "1" || override === "true") return true; - try { - const raw: unknown = JSON.parse(readFileSync(configPath(cwd), "utf8")); - if (!raw || typeof raw !== "object" || Array.isArray(raw)) return true; - const pabcd = (raw as Record).pabcd; - if (!pabcd || typeof pabcd !== "object" || Array.isArray(pabcd)) return true; - return (pabcd as Record).enabled !== false; - } catch { - return true; - } -} - export function runSubagentStopGate(payload: SubagentStopPayload): string { try { if (!readPabcdEnabled(payload.cwd)) return ""; From 7df4c1b33350bb0b5b25d72bfb77542f40493958 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 00:46:38 +0900 Subject: [PATCH 21/90] fix(pabcd-state): ignore runtime state on first directory creation (#255) --- .../components/bg-wake/src/codexclaw-dir.ts | 26 ++++ .../components/cxc-ops/src/codexclaw-dir.ts | 26 ++++ .../messenger-bridge/src/codexclaw-dir.ts | 26 ++++ .../pabcd-state/src/codexclaw-dir.ts | 26 ++++ .../components/pabcd-state/src/state.ts | 6 + .../components/pabcd-state/test/state.test.ts | 112 +++++++++++++++++- .../subagent-config/src/codexclaw-dir.ts | 26 ++++ .../test/codexclaw-dir-copies.test.mjs | 12 ++ 8 files changed, 259 insertions(+), 1 deletion(-) create mode 100644 plugins/codexclaw/components/bg-wake/src/codexclaw-dir.ts create mode 100644 plugins/codexclaw/components/cxc-ops/src/codexclaw-dir.ts create mode 100644 plugins/codexclaw/components/messenger-bridge/src/codexclaw-dir.ts create mode 100644 plugins/codexclaw/components/pabcd-state/src/codexclaw-dir.ts create mode 100644 plugins/codexclaw/components/subagent-config/src/codexclaw-dir.ts create mode 100644 plugins/codexclaw/test/codexclaw-dir-copies.test.mjs diff --git a/plugins/codexclaw/components/bg-wake/src/codexclaw-dir.ts b/plugins/codexclaw/components/bg-wake/src/codexclaw-dir.ts new file mode 100644 index 00000000..995dc0ea --- /dev/null +++ b/plugins/codexclaw/components/bg-wake/src/codexclaw-dir.ts @@ -0,0 +1,26 @@ +import { mkdirSync, rmdirSync, writeFileSync } from "node:fs"; +import { join } from "node:path"; + +export const GITIGNORE_TEXT = "# CodexClaw wrote this when it created .codexclaw; everything here is local runtime state.\n*\n!.gitignore\n!rules/\n!rules/*.md\n"; + +/** Only the process that first creates the project-local root may publish its ignore file. */ +export function ensureCodexclawDir( + cwd: string, + writeIgnore: (path: string, data: string, options: { flag: "wx" }) => void = writeFileSync, +): string { + const dir = join(cwd, ".codexclaw"); + try { + mkdirSync(dir); + } catch (err) { + if ((err as NodeJS.ErrnoException)?.code === "EEXIST") return dir; + throw err; + } + try { + writeIgnore(join(dir, ".gitignore"), GITIGNORE_TEXT, { flag: "wx" }); + } catch (err) { + if ((err as NodeJS.ErrnoException)?.code === "EEXIST") return dir; + try { rmdirSync(dir); } catch { /* A concurrent writer made the directory non-empty. */ } + throw err; + } + return dir; +} diff --git a/plugins/codexclaw/components/cxc-ops/src/codexclaw-dir.ts b/plugins/codexclaw/components/cxc-ops/src/codexclaw-dir.ts new file mode 100644 index 00000000..995dc0ea --- /dev/null +++ b/plugins/codexclaw/components/cxc-ops/src/codexclaw-dir.ts @@ -0,0 +1,26 @@ +import { mkdirSync, rmdirSync, writeFileSync } from "node:fs"; +import { join } from "node:path"; + +export const GITIGNORE_TEXT = "# CodexClaw wrote this when it created .codexclaw; everything here is local runtime state.\n*\n!.gitignore\n!rules/\n!rules/*.md\n"; + +/** Only the process that first creates the project-local root may publish its ignore file. */ +export function ensureCodexclawDir( + cwd: string, + writeIgnore: (path: string, data: string, options: { flag: "wx" }) => void = writeFileSync, +): string { + const dir = join(cwd, ".codexclaw"); + try { + mkdirSync(dir); + } catch (err) { + if ((err as NodeJS.ErrnoException)?.code === "EEXIST") return dir; + throw err; + } + try { + writeIgnore(join(dir, ".gitignore"), GITIGNORE_TEXT, { flag: "wx" }); + } catch (err) { + if ((err as NodeJS.ErrnoException)?.code === "EEXIST") return dir; + try { rmdirSync(dir); } catch { /* A concurrent writer made the directory non-empty. */ } + throw err; + } + return dir; +} diff --git a/plugins/codexclaw/components/messenger-bridge/src/codexclaw-dir.ts b/plugins/codexclaw/components/messenger-bridge/src/codexclaw-dir.ts new file mode 100644 index 00000000..995dc0ea --- /dev/null +++ b/plugins/codexclaw/components/messenger-bridge/src/codexclaw-dir.ts @@ -0,0 +1,26 @@ +import { mkdirSync, rmdirSync, writeFileSync } from "node:fs"; +import { join } from "node:path"; + +export const GITIGNORE_TEXT = "# CodexClaw wrote this when it created .codexclaw; everything here is local runtime state.\n*\n!.gitignore\n!rules/\n!rules/*.md\n"; + +/** Only the process that first creates the project-local root may publish its ignore file. */ +export function ensureCodexclawDir( + cwd: string, + writeIgnore: (path: string, data: string, options: { flag: "wx" }) => void = writeFileSync, +): string { + const dir = join(cwd, ".codexclaw"); + try { + mkdirSync(dir); + } catch (err) { + if ((err as NodeJS.ErrnoException)?.code === "EEXIST") return dir; + throw err; + } + try { + writeIgnore(join(dir, ".gitignore"), GITIGNORE_TEXT, { flag: "wx" }); + } catch (err) { + if ((err as NodeJS.ErrnoException)?.code === "EEXIST") return dir; + try { rmdirSync(dir); } catch { /* A concurrent writer made the directory non-empty. */ } + throw err; + } + return dir; +} diff --git a/plugins/codexclaw/components/pabcd-state/src/codexclaw-dir.ts b/plugins/codexclaw/components/pabcd-state/src/codexclaw-dir.ts new file mode 100644 index 00000000..995dc0ea --- /dev/null +++ b/plugins/codexclaw/components/pabcd-state/src/codexclaw-dir.ts @@ -0,0 +1,26 @@ +import { mkdirSync, rmdirSync, writeFileSync } from "node:fs"; +import { join } from "node:path"; + +export const GITIGNORE_TEXT = "# CodexClaw wrote this when it created .codexclaw; everything here is local runtime state.\n*\n!.gitignore\n!rules/\n!rules/*.md\n"; + +/** Only the process that first creates the project-local root may publish its ignore file. */ +export function ensureCodexclawDir( + cwd: string, + writeIgnore: (path: string, data: string, options: { flag: "wx" }) => void = writeFileSync, +): string { + const dir = join(cwd, ".codexclaw"); + try { + mkdirSync(dir); + } catch (err) { + if ((err as NodeJS.ErrnoException)?.code === "EEXIST") return dir; + throw err; + } + try { + writeIgnore(join(dir, ".gitignore"), GITIGNORE_TEXT, { flag: "wx" }); + } catch (err) { + if ((err as NodeJS.ErrnoException)?.code === "EEXIST") return dir; + try { rmdirSync(dir); } catch { /* A concurrent writer made the directory non-empty. */ } + throw err; + } + return dir; +} diff --git a/plugins/codexclaw/components/pabcd-state/src/state.ts b/plugins/codexclaw/components/pabcd-state/src/state.ts index 066db540..646a263f 100644 --- a/plugins/codexclaw/components/pabcd-state/src/state.ts +++ b/plugins/codexclaw/components/pabcd-state/src/state.ts @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.ts"; import { existsSync, mkdirSync, readFileSync, writeFileSync, appendFileSync, linkSync, rmSync, statSync } from "node:fs"; import { randomUUID } from "node:crypto"; import { isAbsolute, join, resolve } from "node:path"; @@ -374,6 +375,7 @@ export function ensureState( throw new TypeError("sessionId must be a canonical state key"); } const dir = sessionsDir(cwd); + ensureCodexclawDir(cwd); mkdirSync(dir, { recursive: true }); const finalPath = statePath(cwd, sessionId); const tmp = `${finalPath}.${process.pid}.${randomUUID()}.tmp`; @@ -606,6 +608,7 @@ export function readStateStrict(cwd: string, sessionId: string): { state: State; export function writeState(cwd: string, next: State): void { const dir = sessionsDir(cwd); + ensureCodexclawDir(cwd); mkdirSync(dir, { recursive: true }); const finalPath = statePath(cwd, next.sessionId); const tmp = `${finalPath}.${process.pid}.${Date.now()}.tmp`; @@ -653,6 +656,7 @@ function sleepSyncMs(ms: number): void { export function withSessionLock(cwd: string, sessionId: string, fn: () => T): T { const dir = sessionsDir(cwd); + ensureCodexclawDir(cwd); mkdirSync(dir, { recursive: true }); const lockPath = `${statePath(cwd, sessionId)}.lock`; let held = false; @@ -680,6 +684,7 @@ export function withSessionLock(cwd: string, sessionId: string, fn: () => T): export function appendLedger(cwd: string, entry: LedgerEntry): void { const dir = join(cwd, STATE_DIR); + ensureCodexclawDir(cwd); mkdirSync(dir, { recursive: true }); appendFileSync(join(dir, LEDGER_FILE), `${JSON.stringify(entry)}\n`); } @@ -737,6 +742,7 @@ function interviewLedgerPath(cwd: string, sessionId: string): string { */ export function appendInterviewEvent(cwd: string, entry: InterviewEvent): void { const dir = interviewsDir(cwd); + ensureCodexclawDir(cwd); mkdirSync(dir, { recursive: true }); appendFileSync(interviewLedgerPath(cwd, entry.sessionId), `${JSON.stringify(entry)}\n`); } diff --git a/plugins/codexclaw/components/pabcd-state/test/state.test.ts b/plugins/codexclaw/components/pabcd-state/test/state.test.ts index 405ab7c9..011ab037 100644 --- a/plugins/codexclaw/components/pabcd-state/test/state.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/state.test.ts @@ -1,8 +1,11 @@ import { test } from "node:test"; import assert from "node:assert/strict"; -import { mkdtempSync, rmSync, writeFileSync, mkdirSync, readFileSync, existsSync, readdirSync, appendFileSync } from "node:fs"; +import { mkdtempSync, rmSync, writeFileSync, mkdirSync, readFileSync, existsSync, readdirSync, appendFileSync, symlinkSync } from "node:fs"; import { tmpdir } from "node:os"; import { join } from "node:path"; +import { spawn, spawnSync } from "node:child_process"; +import { handleSessionStart } from "../src/hook.ts"; +import { ensureCodexclawDir, GITIGNORE_TEXT } from "../src/codexclaw-dir.ts"; import { readState, writeState, @@ -24,6 +27,113 @@ function freshCwd(): string { return mkdtempSync(join(tmpdir(), "codexclaw-state-")); } +const IGNORE_TEXT = "# CodexClaw wrote this when it created .codexclaw; everything here is local runtime state.\n*\n!.gitignore\n!rules/\n!rules/*.md\n"; + +test("issue255: helper publishes exact bytes and leaves an existing empty folder alone", () => { + const fresh = freshCwd(), existing = freshCwd(); + try { + assert.equal(ensureCodexclawDir(fresh), join(fresh, ".codexclaw")); + assert.equal(readFileSync(join(fresh, ".codexclaw", ".gitignore"), "utf8"), IGNORE_TEXT); + assert.equal(GITIGNORE_TEXT, IGNORE_TEXT); + mkdirSync(join(existing, ".codexclaw")); + ensureCodexclawDir(existing); + assert.equal(existsSync(join(existing, ".codexclaw", ".gitignore")), false); + } finally { rmSync(fresh, { recursive: true, force: true }); rmSync(existing, { recursive: true, force: true }); } +}); + +test("issue255: helper leaves an existing symlink and its target untouched", (t) => { + const cwd = freshCwd(), target = freshCwd(); + try { + try { symlinkSync(target, join(cwd, ".codexclaw"), "dir"); } + catch (err) { + if (["EPERM", "EACCES"].includes((err as NodeJS.ErrnoException).code ?? "")) { t.skip("directory symlinks unavailable"); return; } + throw err; + } + ensureCodexclawDir(cwd); + assert.deepEqual(readdirSync(target), []); + } finally { rmSync(cwd, { recursive: true, force: true }); rmSync(target, { recursive: true, force: true }); } +}); + +test("issue255: identical EEXIST ignore write succeeds without overwriting", () => { + const cwd = freshCwd(); + try { + ensureCodexclawDir(cwd, (path, data) => { + writeFileSync(path, data, { flag: "wx" }); + throw Object.assign(new Error("raced"), { code: "EEXIST" }); + }); + assert.equal(readFileSync(join(cwd, ".codexclaw", ".gitignore"), "utf8"), IGNORE_TEXT); + } finally { rmSync(cwd, { recursive: true, force: true }); } +}); + +test("issue255: conflicting EEXIST ignore write leaves existing bytes alone", () => { + const cwd = freshCwd(); + try { + ensureCodexclawDir(cwd, (path) => { + writeFileSync(path, "other rules\n", { flag: "wx" }); + throw Object.assign(new Error("raced"), { code: "EEXIST" }); + }); + assert.equal(readFileSync(join(cwd, ".codexclaw", ".gitignore"), "utf8"), "other rules\n"); + } finally { rmSync(cwd, { recursive: true, force: true }); } +}); + +test("issue255: ignore write failure removes its empty directory and rethrows", () => { + const cwd = freshCwd(); + try { + assert.throws(() => ensureCodexclawDir(cwd, () => { throw Object.assign(new Error("denied"), { code: "EACCES" }); }), /denied/); + assert.equal(existsSync(join(cwd, ".codexclaw")), false); + } finally { rmSync(cwd, { recursive: true, force: true }); } +}); + +test("issue255: concurrent first SessionStart creators keep valid state and one ignore file", async () => { + const cwd = freshCwd(); + try { + const moduleUrl = new URL("../dist/state.js", import.meta.url).href; + const script = `import { ensureState } from ${JSON.stringify(moduleUrl)}; ensureState(process.argv[1], "concurrent");`; + const run = () => new Promise((resolveExit) => { + const child = spawn(process.execPath, ["--input-type=module", "-e", script, cwd], { stdio: "ignore" }); + child.once("exit", (code) => resolveExit(code)); + }); + assert.deepEqual(await Promise.all([run(), run()]), [0, 0]); + assert.equal(readFileSync(join(cwd, ".codexclaw", ".gitignore"), "utf8"), IGNORE_TEXT); + assert.equal(JSON.parse(readFileSync(join(cwd, ".codexclaw", "sessions", "concurrent.json"), "utf8")).sessionId, "concurrent"); + } finally { rmSync(cwd, { recursive: true, force: true }); } +}); + +test("issue255: SessionStart creates state and exact local ignore file", () => { + const cwd = freshCwd(); + try { + assert.equal(handleSessionStart({ hook_event_name: "SessionStart", cwd, session_id: "issue255" }), ""); + assert.ok(existsSync(join(cwd, ".codexclaw", "sessions", "issue255.json"))); + assert.equal(readFileSync(join(cwd, ".codexclaw", ".gitignore"), "utf8"), IGNORE_TEXT); + } finally { rmSync(cwd, { recursive: true, force: true }); } +}); + +test("issue255: existing state directory and ignore file are untouched", () => { + const cwd = freshCwd(); + try { + mkdirSync(join(cwd, ".codexclaw")); + ensureState(cwd, "existing-empty"); + assert.equal(existsSync(join(cwd, ".codexclaw", ".gitignore")), false); + writeFileSync(join(cwd, ".codexclaw", ".gitignore"), "user rules\n"); + ensureState(cwd, "existing-ignore"); + assert.equal(readFileSync(join(cwd, ".codexclaw", ".gitignore"), "utf8"), "user rules\n"); + } finally { rmSync(cwd, { recursive: true, force: true }); } +}); + +test("issue255: Git ignores runtime state and exposes only project rules without an ancestor ignore", () => { + const cwd = freshCwd(); + try { + assert.equal(spawnSync("git", ["init", "-q", cwd]).status, 0); + ensureState(cwd, "x"); + const run = (path: string) => spawnSync("git", ["-C", cwd, "check-ignore", "-q", path]).status; + for (const path of [".codexclaw/sessions/x.json", ".codexclaw/ledger.jsonl", ".codexclaw/goalplans/slug/goalplan.json"]) assert.equal(run(path), 0, path); + for (const path of [".codexclaw/.gitignore", ".codexclaw/rules/a.md"]) assert.equal(run(path), 1, path); + writeFileSync(join(cwd, ".gitignore"), ".codexclaw/\n"); + assert.equal(run(".codexclaw/rules/a.md"), 0); + assert.equal(readFileSync(join(cwd, ".codexclaw", ".gitignore"), "utf8"), IGNORE_TEXT); + } finally { rmSync(cwd, { recursive: true, force: true }); } +}); + test("SessionStart ensureState: fresh session creates the exact default IDLE state without temp files", () => { const cwd = freshCwd(); try { diff --git a/plugins/codexclaw/components/subagent-config/src/codexclaw-dir.ts b/plugins/codexclaw/components/subagent-config/src/codexclaw-dir.ts new file mode 100644 index 00000000..995dc0ea --- /dev/null +++ b/plugins/codexclaw/components/subagent-config/src/codexclaw-dir.ts @@ -0,0 +1,26 @@ +import { mkdirSync, rmdirSync, writeFileSync } from "node:fs"; +import { join } from "node:path"; + +export const GITIGNORE_TEXT = "# CodexClaw wrote this when it created .codexclaw; everything here is local runtime state.\n*\n!.gitignore\n!rules/\n!rules/*.md\n"; + +/** Only the process that first creates the project-local root may publish its ignore file. */ +export function ensureCodexclawDir( + cwd: string, + writeIgnore: (path: string, data: string, options: { flag: "wx" }) => void = writeFileSync, +): string { + const dir = join(cwd, ".codexclaw"); + try { + mkdirSync(dir); + } catch (err) { + if ((err as NodeJS.ErrnoException)?.code === "EEXIST") return dir; + throw err; + } + try { + writeIgnore(join(dir, ".gitignore"), GITIGNORE_TEXT, { flag: "wx" }); + } catch (err) { + if ((err as NodeJS.ErrnoException)?.code === "EEXIST") return dir; + try { rmdirSync(dir); } catch { /* A concurrent writer made the directory non-empty. */ } + throw err; + } + return dir; +} diff --git a/plugins/codexclaw/test/codexclaw-dir-copies.test.mjs b/plugins/codexclaw/test/codexclaw-dir-copies.test.mjs new file mode 100644 index 00000000..95a4380f --- /dev/null +++ b/plugins/codexclaw/test/codexclaw-dir-copies.test.mjs @@ -0,0 +1,12 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import { fileURLToPath } from "node:url"; +import { dirname, join } from "node:path"; + +test("issue255: project-local directory helpers remain byte-identical across independently built components", () => { + const components = join(dirname(fileURLToPath(import.meta.url)), "..", "components"); + const names = ["pabcd-state", "cxc-ops", "bg-wake", "subagent-config", "messenger-bridge"]; + const copies = names.map((name) => readFileSync(join(components, name, "src", "codexclaw-dir.ts"))); + for (let i = 1; i < copies.length; i++) assert.deepEqual(copies[i], copies[0], names[i]); +}); From dcdb44cb2ed92649929261eb2f7e111b3c24bd15 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 00:46:56 +0900 Subject: [PATCH 22/90] fix(codexclaw): route project-local writers through ignore helper (#255) --- plugins/codexclaw/components/bg-wake/src/store.ts | 2 ++ .../components/bg-wake/test/registry.test.ts | 11 +++++++++-- .../components/cxc-ops/src/activation-trace.ts | 2 ++ .../components/cxc-ops/src/map-affordance.ts | 4 +++- .../cxc-ops/test/compact-affordance.test.ts | 10 +++++++++- .../messenger-bridge/src/bridge-controller.ts | 2 +- .../codexclaw/components/messenger-bridge/src/db.ts | 2 ++ .../components/messenger-bridge/src/event-log.ts | 3 +++ .../components/pabcd-state/src/divergence.ts | 3 +++ .../components/pabcd-state/src/edit-shape.ts | 5 +++-- .../components/pabcd-state/src/freeze-cli.ts | 2 ++ .../codexclaw/components/pabcd-state/src/friction.ts | 5 +++-- .../codexclaw/components/pabcd-state/src/goalplan.ts | 3 +++ .../components/pabcd-state/src/interview-ledger.ts | 2 ++ .../codexclaw/components/pabcd-state/src/metrics.ts | 4 +++- .../components/pabcd-state/src/receipt-cli.ts | 2 ++ .../components/pabcd-state/src/release-cli.ts | 10 ++++++---- .../components/pabcd-state/src/render-observations.ts | 7 ++++--- .../components/pabcd-state/src/rule-impact-ledger.ts | 7 ++++++- .../components/pabcd-state/src/session-source.ts | 2 ++ .../components/pabcd-state/src/subagent-evidence.ts | 4 ++++ .../components/pabcd-state/src/worktree-guard.ts | 2 ++ .../components/pabcd-state/test/friction.test.ts | 10 +++++++++- .../pabcd-state/test/memory-write-gate.test.ts | 6 ++++-- .../pabcd-state/test/session-binding.test.ts | 2 +- .../subagent-config/src/fallback-dispatch.ts | 6 ++++-- .../codexclaw/components/subagent-config/src/store.ts | 10 ++++++---- 27 files changed, 100 insertions(+), 28 deletions(-) diff --git a/plugins/codexclaw/components/bg-wake/src/store.ts b/plugins/codexclaw/components/bg-wake/src/store.ts index c1f44eb8..97c2f023 100644 --- a/plugins/codexclaw/components/bg-wake/src/store.ts +++ b/plugins/codexclaw/components/bg-wake/src/store.ts @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.ts"; /** * store.ts — filesystem substrate for the bg registry. * @@ -44,6 +45,7 @@ export function enabledAtPath(cwd: string): string { export function ensureDir(cwd: string): string { const dir = bgDir(cwd); + ensureCodexclawDir(cwd); mkdirSync(dir, { recursive: true }); return dir; } diff --git a/plugins/codexclaw/components/bg-wake/test/registry.test.ts b/plugins/codexclaw/components/bg-wake/test/registry.test.ts index 8979b56b..0eb5c288 100644 --- a/plugins/codexclaw/components/bg-wake/test/registry.test.ts +++ b/plugins/codexclaw/components/bg-wake/test/registry.test.ts @@ -3,7 +3,7 @@ */ import { test } from "node:test"; import assert from "node:assert/strict"; -import { mkdtempSync, writeFileSync } from "node:fs"; +import { mkdtempSync, writeFileSync, readFileSync, rmSync } from "node:fs"; import { tmpdir } from "node:os"; import { join } from "node:path"; import { reconcile, selectWake, writeRecord, listRecords, type BgRecord } from "../src/registry.ts"; @@ -16,6 +16,14 @@ function workspace(): string { return dir; } +test("issue255: bg-wake first writer creates local ignore file", () => { + const cwd = mkdtempSync(join(tmpdir(), "bgreg-first-")); + try { + ensureDir(cwd); + assert.match(readFileSync(join(cwd, ".codexclaw", ".gitignore"), "utf8"), /^# CodexClaw wrote this/); + } finally { rmSync(cwd, { recursive: true, force: true }); } +}); + function running(cwd: string, id: string, pid: number | null): BgRecord { const rec: BgRecord = { id, sessionId: "S1", adoptedBy: null, cwd, command: ["sleep", "1"], note: null, @@ -95,4 +103,3 @@ test("ids do not collide with existing records", () => { running(cwd, "taken", null); assert.notEqual(newId(cwd, "taken"), "taken"); }); - diff --git a/plugins/codexclaw/components/cxc-ops/src/activation-trace.ts b/plugins/codexclaw/components/cxc-ops/src/activation-trace.ts index a3cfccde..64b0fed1 100644 --- a/plugins/codexclaw/components/cxc-ops/src/activation-trace.ts +++ b/plugins/codexclaw/components/cxc-ops/src/activation-trace.ts @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.ts"; /** * activation-trace.ts — opt-in eval trace recorder (issue #11). * @@ -134,6 +135,7 @@ export function emitTrace( ): string | null { if (!isTracingEnabled(env)) return null; const dir = join(outputDir, ".codexclaw", "traces"); + ensureCodexclawDir(outputDir); mkdirSync(dir, { recursive: true }); const path = join(dir, "activations.jsonl"); writeFileSync(path, JSON.stringify(trace) + "\n", { flag: "a" }); diff --git a/plugins/codexclaw/components/cxc-ops/src/map-affordance.ts b/plugins/codexclaw/components/cxc-ops/src/map-affordance.ts index d8683405..a886db9f 100644 --- a/plugins/codexclaw/components/cxc-ops/src/map-affordance.ts +++ b/plugins/codexclaw/components/cxc-ops/src/map-affordance.ts @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.ts"; /** * map-affordance.ts — SessionStart `cxc map` discoverability injector. * @@ -245,7 +246,8 @@ function recoveryPath(stdin: string, event: string, create: boolean): string | n try { st = lstatSync(path); } catch (error) { if (!create || (error as NodeJS.ErrnoException).code !== "ENOENT") return null; - mkdirSync(path, { mode: 0o700 }); + if (path === state) ensureCodexclawDir(realpathSync(p.cwd)); + else mkdirSync(path, { mode: 0o700 }); st = lstatSync(path); } if (!st.isDirectory() || st.isSymbolicLink()) return null; diff --git a/plugins/codexclaw/components/cxc-ops/test/compact-affordance.test.ts b/plugins/codexclaw/components/cxc-ops/test/compact-affordance.test.ts index 35d78579..dbf2f1f5 100644 --- a/plugins/codexclaw/components/cxc-ops/test/compact-affordance.test.ts +++ b/plugins/codexclaw/components/cxc-ops/test/compact-affordance.test.ts @@ -15,6 +15,14 @@ const payload = (cwd: string, event: string, session = "parent", extra = {}) => JSON.stringify({ cwd, session_id: session, hook_event_name: event, ...extra }); const temp = () => mkdtempSync(join(tmpdir(), "cxc-compact-affordance-")); +test("issue255: PostCompact first writer creates local ignore file", () => { + const cwd = temp(); + try { + assert.equal(affordance.runPostCompactAffordance(payload(cwd, "PostCompact")), ""); + assert.match(readFileSync(join(cwd, ".codexclaw", ".gitignore"), "utf8"), /^# CodexClaw wrote this/); + } finally { rmSync(cwd, { recursive: true, force: true }); } +}); + test("PostCompact returns no unsupported event-specific context", () => { const cwd = temp(); try { assert.equal(affordance.runPostCompactAffordance(payload(cwd, "PostCompact")), ""); } @@ -35,7 +43,7 @@ test("compact hint is emitted once at the next root prompt, without FSM files", assert.match(out.hookSpecificOutput.additionalContext, /User questions:.*request_user_input_async/); assert.equal(out.hookSpecificOutput.additionalContext.split("User questions:").length - 1, 1); assert.equal(affordance.runUserPromptAffordance(payload(cwd, "UserPromptSubmit")), ""); - assert.deepEqual(readdirSync(join(cwd, ".codexclaw")), ["affordance-recovery"]); + assert.deepEqual(readdirSync(join(cwd, ".codexclaw")), [".gitignore", "affordance-recovery"]); assert.deepEqual(readdirSync(dir), []); } finally { rmSync(cwd, { recursive: true, force: true }); } }); diff --git a/plugins/codexclaw/components/messenger-bridge/src/bridge-controller.ts b/plugins/codexclaw/components/messenger-bridge/src/bridge-controller.ts index bdb92f32..46fcbc72 100644 --- a/plugins/codexclaw/components/messenger-bridge/src/bridge-controller.ts +++ b/plugins/codexclaw/components/messenger-bridge/src/bridge-controller.ts @@ -90,7 +90,7 @@ export class BridgeController { this.opts = opts; this.db = opts.db; this.log = opts.log ?? (() => {}); - this.events = new EventLog({ path: join(opts.workdir, ".codexclaw", "bridge-events.jsonl") }); + this.events = new EventLog({ path: join(opts.workdir, ".codexclaw", "bridge-events.jsonl"), projectCwd: opts.workdir }); } /** Shared AgentService accessor (heartbeat scheduler rides the same queues diff --git a/plugins/codexclaw/components/messenger-bridge/src/db.ts b/plugins/codexclaw/components/messenger-bridge/src/db.ts index 5bba2c07..e0763ea0 100644 --- a/plugins/codexclaw/components/messenger-bridge/src/db.ts +++ b/plugins/codexclaw/components/messenger-bridge/src/db.ts @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.ts"; /** * db.ts — bridge state substrate (messenger-bridge Phase 1). * @@ -1036,6 +1037,7 @@ ALTER TABLE agents ADD COLUMN tool_progress TEXT NOT NULL DEFAULT 'new' /** Open (creating if needed) the project-scoped bridge DB with 600 perms. */ export function openBridgeDb(cwd: string): BridgeDb { const dir = join(cwd, ".codexclaw"); + ensureCodexclawDir(cwd); mkdirSync(dir, { recursive: true, mode: 0o700 }); const file = join(dir, "bridge.db"); const db = new BridgeDb(file); diff --git a/plugins/codexclaw/components/messenger-bridge/src/event-log.ts b/plugins/codexclaw/components/messenger-bridge/src/event-log.ts index 6120e73c..0f4cb137 100644 --- a/plugins/codexclaw/components/messenger-bridge/src/event-log.ts +++ b/plugins/codexclaw/components/messenger-bridge/src/event-log.ts @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.ts"; /** Ordered, non-blocking JSONL event log with bounded rotation and memory ring. */ import { chmodSync, mkdirSync, statSync } from "node:fs"; import { appendFile, rename, rm } from "node:fs/promises"; @@ -16,6 +17,7 @@ export type BridgeEvent = export interface EventLogOptions { path: string; + projectCwd?: string; maxSizeBytes?: number; maxFiles?: number; maxPendingBytes?: number; @@ -52,6 +54,7 @@ export class EventLog { this.maxPendingBytes = Math.max(1, opts.maxPendingBytes ?? DEFAULT_MAX_PENDING_BYTES); this.maxPendingEvents = Math.max(1, opts.maxPendingEvents ?? DEFAULT_MAX_PENDING_EVENTS); this.append = opts.append ?? appendFile; + if (opts.projectCwd) ensureCodexclawDir(opts.projectCwd); mkdirSync(dirname(this.filePath), { recursive: true, mode: 0o700 }); try { this.bytes = statSync(this.filePath).size; diff --git a/plugins/codexclaw/components/pabcd-state/src/divergence.ts b/plugins/codexclaw/components/pabcd-state/src/divergence.ts index a7343a00..6b0b7681 100644 --- a/plugins/codexclaw/components/pabcd-state/src/divergence.ts +++ b/plugins/codexclaw/components/pabcd-state/src/divergence.ts @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.ts"; import { appendFileSync, mkdirSync, readFileSync, rmSync, writeFileSync } from "node:fs"; import { join } from "node:path"; import { renameWithRetry } from "./atomic-write.ts"; @@ -120,6 +121,7 @@ export function writeDivergenceMode( reason: input.reason, updatedAt: input.now?.() ?? new Date().toISOString(), }; + ensureCodexclawDir(cwd); mkdirSync(divergenceDir(cwd), { recursive: true }); const finalPath = modePath(cwd, input.sessionId); const tmp = `${finalPath}.${process.pid}.${Date.now()}.tmp`; @@ -191,6 +193,7 @@ export function recordDivergenceCandidate(cwd: string, input: RecordDivergenceCa ...(input.changeClass ? { changeClass: input.changeClass } : {}), ...(input.killedAtPhase ? { killedAtPhase: input.killedAtPhase } : {}), }; + ensureCodexclawDir(cwd); mkdirSync(divergenceDir(cwd), { recursive: true }); appendFileSync(candidatesPath(cwd), `${JSON.stringify(candidate)}\n`); return candidate; diff --git a/plugins/codexclaw/components/pabcd-state/src/edit-shape.ts b/plugins/codexclaw/components/pabcd-state/src/edit-shape.ts index 6f2d76ab..38d3eca4 100644 --- a/plugins/codexclaw/components/pabcd-state/src/edit-shape.ts +++ b/plugins/codexclaw/components/pabcd-state/src/edit-shape.ts @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.ts"; /** * edit-shape.ts — PostToolUse advisory for repeated same-shaped edits (astgrep_active 00). * @@ -20,7 +21,7 @@ * PostToolUse additionalContext envelope parity: omo lsp/src/codex-hook.ts:36-42. */ import { createHash } from "node:crypto"; -import { appendFileSync, mkdirSync, readFileSync } from "node:fs"; +import { appendFileSync, readFileSync } from "node:fs"; import { join } from "node:path"; import type { PostToolUsePayload } from "./hook.ts"; import { splitLines } from "./text-lines.ts"; @@ -163,7 +164,7 @@ function keyStates(rows: EditShapeRow[]): Map { function appendRow(cwd: string, row: EditShapeRow): void { try { - mkdirSync(join(cwd, STATE_DIR), { recursive: true }); + ensureCodexclawDir(cwd); appendFileSync(ledgerPath(cwd), `${JSON.stringify(row)}\n`); } catch { // best-effort; advisory logic already ran on the in-memory state diff --git a/plugins/codexclaw/components/pabcd-state/src/freeze-cli.ts b/plugins/codexclaw/components/pabcd-state/src/freeze-cli.ts index 8c693693..c061ad79 100644 --- a/plugins/codexclaw/components/pabcd-state/src/freeze-cli.ts +++ b/plugins/codexclaw/components/pabcd-state/src/freeze-cli.ts @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.ts"; /** * freeze-cli.ts — runtime wiring for the L10.3 freeze/stale path (HIGH-1/HIGH-4). * @@ -115,6 +116,7 @@ export function runFreeze(args: FreezeCliArgs): string { } if (!args.dryRun) { + ensureCodexclawDir(args.cwd); mkdirSync(join(args.cwd, STATE_DIR, FREEZE_MANIFEST_DIR), { recursive: true }); writeFileSync(manifestPath, JSON.stringify(manifest, null, 2)); } diff --git a/plugins/codexclaw/components/pabcd-state/src/friction.ts b/plugins/codexclaw/components/pabcd-state/src/friction.ts index b2ccd8d3..833d4256 100644 --- a/plugins/codexclaw/components/pabcd-state/src/friction.ts +++ b/plugins/codexclaw/components/pabcd-state/src/friction.ts @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.ts"; /** * friction.ts — project-local friction ledger (lazygap_impl 080.1). * @@ -17,7 +18,7 @@ * (a read/parse error yields no verdict, so callers allow). */ import { createHash } from "node:crypto"; -import { appendFileSync, mkdirSync, readFileSync } from "node:fs"; +import { appendFileSync, readFileSync } from "node:fs"; import { join } from "node:path"; import { splitLines } from "./text-lines.ts"; @@ -124,7 +125,7 @@ export function recordFriction(cwd: string, tool: string, errorText: string): Fr const verdict = verdictForCount(count); const entry: FrictionEntry = { ts: new Date().toISOString(), key, tool, normalized, count, verdict }; try { - mkdirSync(join(cwd, STATE_DIR), { recursive: true }); + ensureCodexclawDir(cwd); appendFileSync(frictionPath(cwd), `${JSON.stringify(entry)}\n`); } catch { // best-effort; the verdict is still meaningful to the caller diff --git a/plugins/codexclaw/components/pabcd-state/src/goalplan.ts b/plugins/codexclaw/components/pabcd-state/src/goalplan.ts index c4d60ad8..7fd75cc7 100644 --- a/plugins/codexclaw/components/pabcd-state/src/goalplan.ts +++ b/plugins/codexclaw/components/pabcd-state/src/goalplan.ts @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.ts"; /** * goalplan.ts — project-local durable goalplan substrate (lazygap_impl 030). * @@ -850,6 +851,7 @@ function firstInvalidField(parsed: unknown): string { export function writeGoalplan(cwd: string, plan: Goalplan): void { validateGoalplanSlug(plan.slug); const dir = goalplanDir(cwd, plan.slug); + ensureCodexclawDir(cwd); mkdirSync(dir, { recursive: true, mode: 0o700 }); // Recheck after creation to close the ordinary pre-existing symlink case. goalplanDir(cwd, plan.slug); @@ -874,6 +876,7 @@ export function appendGoalplanLedger(cwd: string, slug: string, entry: GoalplanL validateGoalplanSlug(slug); if (entry.slug !== slug) throw new Error("goalplan ledger entry slug does not match target slug"); const dir = goalplanDir(cwd, slug); + ensureCodexclawDir(cwd); mkdirSync(dir, { recursive: true, mode: 0o700 }); goalplanDir(cwd, slug); const path = goalplanLedgerPath(cwd, slug); diff --git a/plugins/codexclaw/components/pabcd-state/src/interview-ledger.ts b/plugins/codexclaw/components/pabcd-state/src/interview-ledger.ts index a2a07b6e..198812d7 100644 --- a/plugins/codexclaw/components/pabcd-state/src/interview-ledger.ts +++ b/plugins/codexclaw/components/pabcd-state/src/interview-ledger.ts @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.ts"; /** * interview-ledger.ts — durable interview question/answer capture (L12 / 120, WP4). * @@ -217,6 +218,7 @@ function alreadyRecorded(cwd: string, sessionId: string, eventId: string): boole function appendEvent(cwd: string, entry: InterviewQaEvent): void { const dir = join(cwd, STATE_DIR, INTERVIEWS_SUBDIR); + ensureCodexclawDir(cwd); mkdirSync(dir, { recursive: true }); appendFileSync(ledgerPath(cwd, entry.sessionId), `${JSON.stringify(entry)}\n`); } diff --git a/plugins/codexclaw/components/pabcd-state/src/metrics.ts b/plugins/codexclaw/components/pabcd-state/src/metrics.ts index 59c2d8a8..a5c468c7 100644 --- a/plugins/codexclaw/components/pabcd-state/src/metrics.ts +++ b/plugins/codexclaw/components/pabcd-state/src/metrics.ts @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.ts"; import { appendFileSync, mkdirSync, readFileSync, rmSync, writeFileSync } from "node:fs"; import { join } from "node:path"; import { renameWithRetry } from "./atomic-write.ts"; @@ -129,12 +130,13 @@ export function recordObjectiveMetric(cwd: string, input: RecordObjectiveMetricI best: Math.max(previousBest, input.value), source: input.source, }; - mkdirSync(codexclawDir(cwd), { recursive: true }); + ensureCodexclawDir(cwd); appendFileSync(metricsPath(cwd), `${JSON.stringify(next)}\n`); return next; } export function writeObjectiveKind(cwd: string, sessionId: string, kind: ObjectiveKind): void { + ensureCodexclawDir(cwd); mkdirSync(objectiveKindDir(cwd), { recursive: true }); const finalPath = objectiveKindPath(cwd, sessionId); const tmp = `${finalPath}.${process.pid}.${Date.now()}.tmp`; diff --git a/plugins/codexclaw/components/pabcd-state/src/receipt-cli.ts b/plugins/codexclaw/components/pabcd-state/src/receipt-cli.ts index 40d256bd..908a68fb 100644 --- a/plugins/codexclaw/components/pabcd-state/src/receipt-cli.ts +++ b/plugins/codexclaw/components/pabcd-state/src/receipt-cli.ts @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.ts"; /** * receipt-cli.ts — `cxc receipt test` (075). * @@ -178,6 +179,7 @@ export function runReceiptCli(args: ReceiptCliArgs): ReceiptCliResult { // Recorded so a reader can see WHICH rewrites the check was allowed to make. ...(args.generated && args.generated.length > 0 ? { generatedPaths: args.generated } : {}), }; + ensureCodexclawDir(args.cwd); mkdirSync(join(path, ".."), { recursive: true }); writeFileSync(path, `${JSON.stringify(receipt, null, 2)}\n`); return { output: path, code: 0 }; diff --git a/plugins/codexclaw/components/pabcd-state/src/release-cli.ts b/plugins/codexclaw/components/pabcd-state/src/release-cli.ts index 44a4cc61..babc769d 100644 --- a/plugins/codexclaw/components/pabcd-state/src/release-cli.ts +++ b/plugins/codexclaw/components/pabcd-state/src/release-cli.ts @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.ts"; /** * release-cli.ts — `cxc release`: assemble a candidate manifest from real receipts * and refuse publication when the evidence does not describe the candidate commit. @@ -131,7 +132,8 @@ function readCandidate(path: string): { manifest: CandidateManifest } | { error: } /** Atomic write so a crashed step never leaves a half-written candidate. */ -function writeCandidate(path: string, manifest: CandidateManifest): void { +function writeCandidate(path: string, manifest: CandidateManifest, projectCwd?: string): void { + if (projectCwd) ensureCodexclawDir(projectCwd); mkdirSync(dirname(path), { recursive: true }); const tmp = path + ".tmp"; writeFileSync(tmp, JSON.stringify(manifest, null, 2) + "\n"); @@ -177,7 +179,7 @@ function runInit(argv: string[], cwd: string): ReleaseCliResult { scorecard: {}, nonGoals: [], }; - writeCandidate(path, manifest); + writeCandidate(path, manifest, explicit ? undefined : cwd); return { code: 0, output: @@ -196,7 +198,7 @@ function mutate( if ("error" in read) return { code: 1, output: "release: " + read.error }; const outcome = fn(read.manifest); if (typeof outcome !== "string") return outcome; - writeCandidate(resolved.path, read.manifest); + writeCandidate(resolved.path, read.manifest, argv.includes("--candidate") ? undefined : cwd); return { code: 0, output: outcome }; } @@ -329,7 +331,7 @@ function runVerify(argv: string[], cwd: string): ReleaseCliResult { // skipped, rather than leaving it in a CI log nobody reads. if (allowDeferred && manifest.allowedDeferred !== true) { manifest.allowedDeferred = true; - writeCandidate(resolved.path, manifest); + writeCandidate(resolved.path, manifest, argv.includes("--candidate") ? undefined : cwd); } if (asJson) { diff --git a/plugins/codexclaw/components/pabcd-state/src/render-observations.ts b/plugins/codexclaw/components/pabcd-state/src/render-observations.ts index 3a3464ad..5f4b7b27 100644 --- a/plugins/codexclaw/components/pabcd-state/src/render-observations.ts +++ b/plugins/codexclaw/components/pabcd-state/src/render-observations.ts @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.ts"; /** * render-observations.ts — L2 render-observation ledger (C-RENDER-GROUNDING-01). * @@ -18,7 +19,7 @@ * All IO is project-local under `cwd`. Every reader FAILS-OPEN (missing file or * parse error yields []). */ -import { appendFileSync, mkdirSync, readFileSync, readdirSync, statSync, writeFileSync } from "node:fs"; +import { appendFileSync, readFileSync, readdirSync, statSync, writeFileSync } from "node:fs"; import { dirname, extname, join, resolve } from "node:path"; import type { PostToolUsePayload } from "./hook.ts"; import { fileEditShapes } from "./edit-shape.ts"; @@ -69,7 +70,7 @@ function ledgerPath(cwd: string): string { function appendRow(cwd: string, row: RenderObsRow): void { try { - mkdirSync(join(cwd, STATE_DIR), { recursive: true }); + ensureCodexclawDir(cwd); appendFileSync(ledgerPath(cwd), `${JSON.stringify(row)}\n`); } catch { // best-effort; advisory logic runs on in-memory state @@ -124,7 +125,7 @@ export function readRenderObsRows(cwd: string): RenderObsRow[] { */ export function resetRenderLedger(cwd: string): void { try { - mkdirSync(join(cwd, STATE_DIR), { recursive: true }); + ensureCodexclawDir(cwd); writeFileSync(ledgerPath(cwd), ""); } catch { // best-effort diff --git a/plugins/codexclaw/components/pabcd-state/src/rule-impact-ledger.ts b/plugins/codexclaw/components/pabcd-state/src/rule-impact-ledger.ts index f08ae3f6..3674dabe 100644 --- a/plugins/codexclaw/components/pabcd-state/src/rule-impact-ledger.ts +++ b/plugins/codexclaw/components/pabcd-state/src/rule-impact-ledger.ts @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.ts"; /** * rule-impact-ledger.ts — opt-in Rule Impact Ledger (issue #18). * @@ -5,7 +6,7 @@ * outcome change, and cost so the router diet can be evidence-based. */ import { writeFileSync, readFileSync, existsSync, mkdirSync } from "node:fs"; -import { join, dirname } from "node:path"; +import { join, dirname, resolve, sep } from "node:path"; /** Schema version for the rule impact ledger. */ export const LEDGER_SCHEMA_VERSION = 1; @@ -102,6 +103,10 @@ export function appendRecord(path: string, record: RuleImpactRecord): void { const ledger = readLedger(path); ledger.records.push(record); const dir = dirname(path); + const absolute = resolve(path); + const marker = `${sep}.codexclaw${sep}`; + const index = absolute.indexOf(marker); + if (index >= 0) ensureCodexclawDir(absolute.slice(0, index)); if (!existsSync(dir)) mkdirSync(dir, { recursive: true }); writeFileSync(path, JSON.stringify(ledger, null, 2)); } diff --git a/plugins/codexclaw/components/pabcd-state/src/session-source.ts b/plugins/codexclaw/components/pabcd-state/src/session-source.ts index c379e640..a3254e25 100644 --- a/plugins/codexclaw/components/pabcd-state/src/session-source.ts +++ b/plugins/codexclaw/components/pabcd-state/src/session-source.ts @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.ts"; /** Immutable per-session source binding; native state/identity never moves. */ import { execFileSync } from "node:child_process"; import { randomUUID } from "node:crypto"; @@ -180,6 +181,7 @@ export function bindSessionSource(cwd: string, sessionId: string, target: string } const binding: SourceBinding = { version: 1, ownerSessionId: sessionId, nativeCwd, sourceRoot, commonDir: source.commonDir, gitDir: source.gitDir }; const path = bindingPath(cwd, sessionId); + ensureCodexclawDir(cwd); mkdirSync(join(cwd, ".codexclaw", "sources"), { recursive: true }); bindingPath(cwd, sessionId); const tmp = `${path}.${randomUUID()}.tmp`; diff --git a/plugins/codexclaw/components/pabcd-state/src/subagent-evidence.ts b/plugins/codexclaw/components/pabcd-state/src/subagent-evidence.ts index ac91f52a..007e43bd 100644 --- a/plugins/codexclaw/components/pabcd-state/src/subagent-evidence.ts +++ b/plugins/codexclaw/components/pabcd-state/src/subagent-evidence.ts @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.ts"; /** * subagent-evidence.ts — SubagentStop evidence-receipt gate (lazygap_impl 010). * @@ -214,6 +215,7 @@ export function readAttempts(cwd: string, sessionId: string, agentId: string, tu export function writeAttempts(cwd: string, sessionId: string, agentId: string, attempts: number, turnId = ""): boolean { try { const p = attemptsPath(cwd, sessionId, agentId, turnId); + ensureCodexclawDir(cwd); mkdirSync(join(cwd, STATE_DIR, EVIDENCE_ATTEMPTS_SUBDIR), { recursive: true }); const tmp = `${p}.${process.pid}.${Date.now()}.tmp`; writeFileSync(tmp, `${JSON.stringify({ attempts })}\n`); @@ -335,6 +337,7 @@ function unrecordableDir(cwd: string): string { */ export function writeUnrecordableMarker(cwd: string, sessionId: string, agentId: string): void { const dir = unrecordableDir(cwd); + ensureCodexclawDir(cwd); mkdirSync(dir, { recursive: true }); const p = join(dir, `${sanitizeKey(sessionId)}-${sanitizeKey(agentId)}-${Date.now()}.json`); writeFileSync(p, `${JSON.stringify({ sessionId, agentId, at: new Date().toISOString() })}\n`, { flag: "wx" }); @@ -350,6 +353,7 @@ export function writeUnrecordableMarker(cwd: string, sessionId: string, agentId: function markerDirWritable(cwd: string): boolean { const probe = join(unrecordableDir(cwd), `.probe-${process.pid}-${Date.now()}`); try { + ensureCodexclawDir(cwd); mkdirSync(unrecordableDir(cwd), { recursive: true }); writeFileSync(probe, "", { flag: "wx" }); rmSync(probe, { force: true }); diff --git a/plugins/codexclaw/components/pabcd-state/src/worktree-guard.ts b/plugins/codexclaw/components/pabcd-state/src/worktree-guard.ts index 3efad2a3..d542a87e 100644 --- a/plugins/codexclaw/components/pabcd-state/src/worktree-guard.ts +++ b/plugins/codexclaw/components/pabcd-state/src/worktree-guard.ts @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.ts"; /** * worktree-guard.ts — Codex-app managed-worktree identity guard (260804 unit, * devlog/_plan/260804_worktree_identity_guardian/010 rev2). @@ -524,6 +525,7 @@ function alreadyInjected(cwd: string, sessionId: string): boolean { function markInjected(cwd: string, sessionId: string, slot: string | null): void { try { const path = markerPath(cwd, sessionId); + ensureCodexclawDir(cwd); mkdirSync(resolve(path, ".."), { recursive: true }); writeFileSync(path, JSON.stringify({ injectedAt: new Date().toISOString(), slot }), { encoding: "utf8", diff --git a/plugins/codexclaw/components/pabcd-state/test/friction.test.ts b/plugins/codexclaw/components/pabcd-state/test/friction.test.ts index df400029..9a4199d0 100644 --- a/plugins/codexclaw/components/pabcd-state/test/friction.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/friction.test.ts @@ -1,6 +1,6 @@ import { test } from "node:test"; import assert from "node:assert/strict"; -import { mkdtempSync, existsSync } from "node:fs"; +import { mkdtempSync, existsSync, readFileSync, rmSync } from "node:fs"; import { tmpdir } from "node:os"; import { join } from "node:path"; import { @@ -18,6 +18,14 @@ function tmp(): string { return mkdtempSync(join(tmpdir(), "cxc-friction-")); } +test("issue255: friction as first PABCD writer creates local ignore file", () => { + const cwd = tmp(); + try { + assert.equal(recordFriction(cwd, "Bash", "first error"), "retry"); + assert.match(readFileSync(join(cwd, ".codexclaw", ".gitignore"), "utf8"), /^# CodexClaw wrote this/); + } finally { rmSync(cwd, { recursive: true, force: true }); } +}); + test("080.1: verdict math retry(1)/escalate(>=2)/stop(>=3)", () => { assert.equal(verdictForCount(1), "retry"); assert.equal(verdictForCount(2), "escalate"); diff --git a/plugins/codexclaw/components/pabcd-state/test/memory-write-gate.test.ts b/plugins/codexclaw/components/pabcd-state/test/memory-write-gate.test.ts index 52b450b3..32c06fe9 100644 --- a/plugins/codexclaw/components/pabcd-state/test/memory-write-gate.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/memory-write-gate.test.ts @@ -9,7 +9,7 @@ */ import { test } from "node:test"; import assert from "node:assert/strict"; -import { mkdtempSync, rmSync } from "node:fs"; +import { mkdtempSync, mkdirSync, rmSync } from "node:fs"; import { homedir, tmpdir } from "node:os"; import { join, resolve } from "node:path"; import { @@ -31,7 +31,9 @@ const SESSION = "019f9d73-4c28-7723-ab52-346aca1d9bcb"; function scratch(): { cwd: string; home: string; env: NodeJS.ProcessEnv } { const dir = mkdtempSync(join(tmpdir(), "cxc-memgate-")); const home = join(dir, "codex-home"); - return { cwd: join(dir, "work"), home, env: { CODEX_HOME: home } }; + const cwd = join(dir, "work"); + mkdirSync(cwd); + return { cwd, home, env: { CODEX_HOME: home } }; } function ptu(overrides: Record = {}): string { diff --git a/plugins/codexclaw/components/pabcd-state/test/session-binding.test.ts b/plugins/codexclaw/components/pabcd-state/test/session-binding.test.ts index 77ffda01..7423f4c3 100644 --- a/plugins/codexclaw/components/pabcd-state/test/session-binding.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/session-binding.test.ts @@ -104,7 +104,7 @@ test("current with no state directory is read-only; bind creates only the child assert.equal(readFileSync(f.path, "utf8"), resumed); assert.deepEqual(readFileSync(parentPath), parent); } - assert.deepEqual(readdirSync(join(f.cwd, ".codexclaw")), ["sessions"]); + assert.deepEqual(readdirSync(join(f.cwd, ".codexclaw")), [".gitignore", "sessions"]); assert.deepEqual(readdirSync(f.dir).sort(), [`${CHILD}.json`, `${PARENT}.json`]); }); diff --git a/plugins/codexclaw/components/subagent-config/src/fallback-dispatch.ts b/plugins/codexclaw/components/subagent-config/src/fallback-dispatch.ts index 8f03f3e4..979088e6 100644 --- a/plugins/codexclaw/components/subagent-config/src/fallback-dispatch.ts +++ b/plugins/codexclaw/components/subagent-config/src/fallback-dispatch.ts @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.ts"; /** Main-owned native spawn protocol. This module selects attempts; it never calls a provider. */ import { randomUUID } from "node:crypto"; import { existsSync, lstatSync, mkdirSync, readFileSync, realpathSync, rmSync, writeFileSync } from "node:fs"; @@ -68,8 +69,9 @@ function taskFailure(raw: unknown): TaskFailure { return { kind: t.kind as TaskFailure["kind"], evidence: smallText(t.evidence, "taskFailure evidence") }; } function directory(cwd: string, sessionId: string): string { - let dir = cwd; - for (const part of [".codexclaw", "dispatches", sessionId]) { + ensureCodexclawDir(cwd); + let dir = join(cwd, ".codexclaw"); + for (const part of ["dispatches", sessionId]) { dir = join(dir, part); if (existsSync(dir)) { if (!lstatSync(dir).isDirectory() || lstatSync(dir).isSymbolicLink()) throw new Error("dispatch directory must not be a symlink"); diff --git a/plugins/codexclaw/components/subagent-config/src/store.ts b/plugins/codexclaw/components/subagent-config/src/store.ts index 394a8429..864d38f4 100644 --- a/plugins/codexclaw/components/subagent-config/src/store.ts +++ b/plugins/codexclaw/components/subagent-config/src/store.ts @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.ts"; /** * store.ts — `.codexclaw/subagents.json` config store (L24 / 240-242). * @@ -249,7 +250,8 @@ export function validateRolePatch(patch: RolePatch): string | null { } /** Atomic write with an exclusive temporary file; preserve unrelated JSON fields. */ -function writeRaw(path: string, config: unknown): void { +function writeRaw(path: string, config: unknown, projectCwd?: string): void { + if (projectCwd) ensureCodexclawDir(projectCwd); mkdirSync(dirname(path), { recursive: true, mode: 0o700 }); const tmp = `${path}.${randomUUID()}.tmp`; try { @@ -262,7 +264,7 @@ function writeRaw(path: string, config: unknown): void { /** Explicit full-config writes remain available to existing callers. */ export function writeConfig(cwd: string, config: SubagentsConfig): void { - writeRaw(storePath(cwd), config); + writeRaw(storePath(cwd), config, cwd); } /** Merge only the selected role; missing roles continue to inherit dynamically. */ @@ -283,7 +285,7 @@ export function setRole(cwd: string, role: RoleName, patch: RolePatch, scope: Co if (next.mode === "default") next.model = null; if (next.fallback) next.fallback = { model: next.fallback.model, effort: next.fallback.effort }; raw.roles[role] = { ...(typeof raw.roles[role] === "object" && raw.roles[role] !== null ? raw.roles[role] as Record : {}), ...next }; - writeRaw(path, raw); + writeRaw(path, raw, scope === "project" ? cwd : undefined); return readConfig(cwd, scope, env); } @@ -294,7 +296,7 @@ export function resetRole(cwd: string, role: RoleName, scope: ConfigScope = "pro const raw = scope === "global" ? readGlobalRaw(env, true) : readRaw(path, true); if (Object.hasOwn(raw.roles, role)) { delete raw.roles[role]; - writeRaw(path, raw); + writeRaw(path, raw, scope === "project" ? cwd : undefined); } return readConfig(cwd, scope, env); } From f0d59563722fd11c70f70096994f2a81917b100b Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 00:54:09 +0900 Subject: [PATCH 23/90] test(pabcd-state): #250 checks ignore #254 turn-budget bookkeeping --- .../components/pabcd-state/test/hook.test.ts | 13 ++++++++++--- 1 file changed, 10 insertions(+), 3 deletions(-) diff --git a/plugins/codexclaw/components/pabcd-state/test/hook.test.ts b/plugins/codexclaw/components/pabcd-state/test/hook.test.ts index 1d398f9d..e7587fac 100644 --- a/plugins/codexclaw/components/pabcd-state/test/hook.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/hook.test.ts @@ -103,6 +103,13 @@ const ISSUE_250_PROMPTS = [ ["neg_plain", "list the files in out/"], ] as const; +// #254 stamps the per-turn Stop budget on every genuine prompt; #250 is about +// PABCD context and arming, so compare everything except that bookkeeping. +function withoutTurnBudget(state: ReturnType) { + const { stopBlockTurnId: _turn, updatedAt: _at, ...rest } = state; + return rest; +} + test("issue 250: nine reported prompts stay silent", () => { for (const [label, prompt] of ISSUE_250_PROMPTS) { assert.equal(detectTrigger(prompt), null, label); @@ -120,7 +127,7 @@ test("issue 250: incidental prompt emits no context and does not arm", () => { const after = readState(cwd, label); assert.equal(after.loopArmSeen, false, label); assert.deepEqual(after.injectedTurns, state.injectedTurns, label); - assert.deepEqual(after, state, label); + assert.deepEqual(withoutTurnBudget(after), withoutTurnBudget(state), label); } finally { rmSync(cwd, { recursive: true, force: true }); } } }); @@ -160,7 +167,7 @@ test("issue 250: inline quoted requests are data", () => { writeState(cwd, defaultState("quoted")); const state = readState(cwd, "quoted"); assert.equal(handleUserPromptSubmit(ups(prompt, cwd, "quoted", "t1")), "", prompt); - assert.deepEqual(readState(cwd, "quoted"), state, prompt); + assert.deepEqual(withoutTurnBudget(readState(cwd, "quoted")), withoutTurnBudget(state), prompt); } finally { rmSync(cwd, { recursive: true, force: true }); } } }); @@ -263,7 +270,7 @@ test("wp3: ordinary Korean C2 remains silent without a CodexClaw request", () => assert.equal(detectTrigger(WP3_ORIGINAL_C2_PROMPT), null); assert.equal(detectLoopArmRequest(WP3_ORIGINAL_C2_PROMPT), false); assert.equal(handleUserPromptSubmit(ups(WP3_ORIGINAL_C2_PROMPT, cwd, session, turn)), ""); - assert.deepEqual(readState(cwd, session), before); + assert.deepEqual(withoutTurnBudget(readState(cwd, session)), withoutTurnBudget(before)); assert.equal(existsSync(join(cwd, STATE_DIR, LEDGER_FILE)), false); } finally { rmSync(cwd, { recursive: true, force: true }); } } From c058adf8c00b7e5f22ee924457a7af621c483281 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 01:04:24 +0900 Subject: [PATCH 24/90] build: regenerate dist for wp2 hook runtime fixes --- .../components/bg-wake/dist/store.js | 2 + .../cxc-ops/dist/activation-trace.js | 2 + .../components/cxc-ops/dist/map-affordance.js | 4 +- .../dist/bridge-controller.js | 2 +- .../components/messenger-bridge/dist/db.js | 2 + .../messenger-bridge/dist/event-log.js | 3 + .../components/pabcd-state/dist/cli.js | 23 ++- .../components/pabcd-state/dist/divergence.js | 3 + .../components/pabcd-state/dist/edit-shape.js | 5 +- .../components/pabcd-state/dist/freeze-cli.js | 2 + .../components/pabcd-state/dist/friction.js | 5 +- .../components/pabcd-state/dist/goal-gate.js | 13 +- .../components/pabcd-state/dist/goalplan.js | 3 + .../components/pabcd-state/dist/hook.js | 153 ++++++++++-------- .../pabcd-state/dist/interview-ledger.js | 2 + .../pabcd-state/dist/interview-policy.js | 14 ++ .../components/pabcd-state/dist/metrics.js | 4 +- .../pabcd-state/dist/receipt-cli.js | 2 + .../pabcd-state/dist/release-cli.js | 10 +- .../pabcd-state/dist/render-observations.js | 7 +- .../pabcd-state/dist/rule-impact-ledger.js | 7 +- .../pabcd-state/dist/session-source.js | 2 + .../components/pabcd-state/dist/state.js | 19 ++- .../pabcd-state/dist/subagent-evidence.js | 16 +- .../pabcd-state/dist/worktree-guard.js | 2 + .../subagent-config/dist/fallback-dispatch.js | 6 +- .../subagent-config/dist/spawn-wrapper.js | 2 + .../components/subagent-config/dist/store.js | 10 +- 28 files changed, 228 insertions(+), 97 deletions(-) diff --git a/plugins/codexclaw/components/bg-wake/dist/store.js b/plugins/codexclaw/components/bg-wake/dist/store.js index 04a46c85..23afa00e 100644 --- a/plugins/codexclaw/components/bg-wake/dist/store.js +++ b/plugins/codexclaw/components/bg-wake/dist/store.js @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.js"; /** * store.ts — filesystem substrate for the bg registry. * @@ -44,6 +45,7 @@ export function enabledAtPath(cwd ) { export function ensureDir(cwd ) { const dir = bgDir(cwd); + ensureCodexclawDir(cwd); mkdirSync(dir, { recursive: true }); return dir; } diff --git a/plugins/codexclaw/components/cxc-ops/dist/activation-trace.js b/plugins/codexclaw/components/cxc-ops/dist/activation-trace.js index d1a7ac04..261c5e64 100644 --- a/plugins/codexclaw/components/cxc-ops/dist/activation-trace.js +++ b/plugins/codexclaw/components/cxc-ops/dist/activation-trace.js @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.js"; /** * activation-trace.ts — opt-in eval trace recorder (issue #11). * @@ -134,6 +135,7 @@ export function emitTrace( ) { if (!isTracingEnabled(env)) return null; const dir = join(outputDir, ".codexclaw", "traces"); + ensureCodexclawDir(outputDir); mkdirSync(dir, { recursive: true }); const path = join(dir, "activations.jsonl"); writeFileSync(path, JSON.stringify(trace) + "\n", { flag: "a" }); diff --git a/plugins/codexclaw/components/cxc-ops/dist/map-affordance.js b/plugins/codexclaw/components/cxc-ops/dist/map-affordance.js index 199fc959..de7c5dd6 100644 --- a/plugins/codexclaw/components/cxc-ops/dist/map-affordance.js +++ b/plugins/codexclaw/components/cxc-ops/dist/map-affordance.js @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.js"; /** * map-affordance.ts — SessionStart `cxc map` discoverability injector. * @@ -245,7 +246,8 @@ function recoveryPath(stdin , event , create ) try { st = lstatSync(path); } catch (error) { if (!create || (error ).code !== "ENOENT") return null; - mkdirSync(path, { mode: 0o700 }); + if (path === state) ensureCodexclawDir(realpathSync(p.cwd)); + else mkdirSync(path, { mode: 0o700 }); st = lstatSync(path); } if (!st.isDirectory() || st.isSymbolicLink()) return null; diff --git a/plugins/codexclaw/components/messenger-bridge/dist/bridge-controller.js b/plugins/codexclaw/components/messenger-bridge/dist/bridge-controller.js index dd936ea0..b380a3d7 100644 --- a/plugins/codexclaw/components/messenger-bridge/dist/bridge-controller.js +++ b/plugins/codexclaw/components/messenger-bridge/dist/bridge-controller.js @@ -90,7 +90,7 @@ export class BridgeController { this.opts = opts; this.db = opts.db; this.log = opts.log ?? (() => {}); - this.events = new EventLog({ path: join(opts.workdir, ".codexclaw", "bridge-events.jsonl") }); + this.events = new EventLog({ path: join(opts.workdir, ".codexclaw", "bridge-events.jsonl"), projectCwd: opts.workdir }); } /** Shared AgentService accessor (heartbeat scheduler rides the same queues diff --git a/plugins/codexclaw/components/messenger-bridge/dist/db.js b/plugins/codexclaw/components/messenger-bridge/dist/db.js index 82b5f233..a2bc3d72 100644 --- a/plugins/codexclaw/components/messenger-bridge/dist/db.js +++ b/plugins/codexclaw/components/messenger-bridge/dist/db.js @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.js"; /** * db.ts — bridge state substrate (messenger-bridge Phase 1). * @@ -1036,6 +1037,7 @@ ALTER TABLE agents ADD COLUMN tool_progress TEXT NOT NULL DEFAULT 'new' /** Open (creating if needed) the project-scoped bridge DB with 600 perms. */ export function openBridgeDb(cwd ) { const dir = join(cwd, ".codexclaw"); + ensureCodexclawDir(cwd); mkdirSync(dir, { recursive: true, mode: 0o700 }); const file = join(dir, "bridge.db"); const db = new BridgeDb(file); diff --git a/plugins/codexclaw/components/messenger-bridge/dist/event-log.js b/plugins/codexclaw/components/messenger-bridge/dist/event-log.js index 129094a1..34627d7d 100644 --- a/plugins/codexclaw/components/messenger-bridge/dist/event-log.js +++ b/plugins/codexclaw/components/messenger-bridge/dist/event-log.js @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.js"; /** Ordered, non-blocking JSONL event log with bounded rotation and memory ring. */ import { chmodSync, mkdirSync, statSync } from "node:fs"; import { appendFile, rename, rm } from "node:fs/promises"; @@ -22,6 +23,7 @@ import { dirname } from "node:path"; + const DEFAULT_MAX_SIZE = 50 * 1024 * 1024; @@ -52,6 +54,7 @@ export class EventLog { this.maxPendingBytes = Math.max(1, opts.maxPendingBytes ?? DEFAULT_MAX_PENDING_BYTES); this.maxPendingEvents = Math.max(1, opts.maxPendingEvents ?? DEFAULT_MAX_PENDING_EVENTS); this.append = opts.append ?? appendFile; + if (opts.projectCwd) ensureCodexclawDir(opts.projectCwd); mkdirSync(dirname(this.filePath), { recursive: true, mode: 0o700 }); try { this.bytes = statSync(this.filePath).size; diff --git a/plugins/codexclaw/components/pabcd-state/dist/cli.js b/plugins/codexclaw/components/pabcd-state/dist/cli.js index 42ea028d..79eab444 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/cli.js +++ b/plugins/codexclaw/components/pabcd-state/dist/cli.js @@ -51,6 +51,14 @@ import { handleIdleEditAdvisory } from "./idle-edit.js"; import { handleMemoryWriteGate } from "./memory-write-gate.js"; import { handleAutomationOwnershipGate } from "./automation-ownership-gate.js"; import { handleReviewObserver } from "./review-observer.js"; +import { readPabcdEnabled } from "./interview-policy.js"; + +const PABCD_DISABLED_EVENTS = new Set([ + "session-start", "stop", "post-compact", "post-tool-use", + "subagent-stop", "subagent-stop-review", "pre-tool-use-idle-edit", + "pre-tool-use-friction", "post-tool-use-friction", "post-tool-use-edit-shape", + "post-tool-use-render-observation", +]); // wp10 (090 trim 4c): the ten terminal-only verb modules below are loaded with // dynamic import() inside their own branch instead of at module scope. @@ -392,6 +400,17 @@ async function main() { process.exit(0); } + let hookCwd = process.cwd(); + try { + const payload = JSON.parse(raw); + if (payload && typeof payload === "object" && !Array.isArray(payload)) { + const candidate = (payload ).cwd; + if (typeof candidate === "string" && candidate.length > 0) hookCwd = candidate; + } + } catch { /* malformed hook input keeps process cwd */ } + const pabcdEnabled = readPabcdEnabled(hookCwd); + if (!pabcdEnabled && PABCD_DISABLED_EVENTS.has(event)) process.exit(0); + // pre-tool-use is handled by a dedicated FAIL-CLOSED dispatcher: a thrown // error on a request_user_input call must DENY (R-9), never fail open. It is // outside the generic fail-open try below so the swallow cannot reopen the @@ -409,7 +428,7 @@ async function main() { if (payload) output = handleSessionStart(payload); // side-effect only; always "" } else if (event === "user-prompt-submit") { const payload = parseUserPromptSubmit(raw); - if (payload) output = handleUserPromptSubmit(payload); + if (payload) output = handleUserPromptSubmit(payload, process.platform, {}, { pabcdEnabled }); } else if (event === "stop") { const payload = parseStop(raw); if (payload) output = handleStop(payload); @@ -439,7 +458,7 @@ async function main() { // lint (deny-capable) first; a lint deny wins; otherwise the IDLE-edit advisory // may inject context. Both legs FAIL-OPEN; a crash must never deny the edit. output = handleApplyPatchLint(raw); - if (output === "") output = handleIdleEditAdvisory(raw); + if (pabcdEnabled && output === "") output = handleIdleEditAdvisory(raw); } else if (event === "pre-tool-use-idle-edit") { // 260714 wp3: FAIL-OPEN IDLE-edit advisory (IDLE-EDIT-ADVISORY-01). Allow + // additionalContext only; a crash here must never deny an edit. diff --git a/plugins/codexclaw/components/pabcd-state/dist/divergence.js b/plugins/codexclaw/components/pabcd-state/dist/divergence.js index b9df9405..91753c36 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/divergence.js +++ b/plugins/codexclaw/components/pabcd-state/dist/divergence.js @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.js"; import { appendFileSync, mkdirSync, readFileSync, rmSync, writeFileSync } from "node:fs"; import { join } from "node:path"; import { renameWithRetry } from "./atomic-write.js"; @@ -120,6 +121,7 @@ export function writeDivergenceMode( reason: input.reason, updatedAt: input.now?.() ?? new Date().toISOString(), }; + ensureCodexclawDir(cwd); mkdirSync(divergenceDir(cwd), { recursive: true }); const finalPath = modePath(cwd, input.sessionId); const tmp = `${finalPath}.${process.pid}.${Date.now()}.tmp`; @@ -191,6 +193,7 @@ export function recordDivergenceCandidate(cwd , input ...(input.changeClass ? { changeClass: input.changeClass } : {}), ...(input.killedAtPhase ? { killedAtPhase: input.killedAtPhase } : {}), }; + ensureCodexclawDir(cwd); mkdirSync(divergenceDir(cwd), { recursive: true }); appendFileSync(candidatesPath(cwd), `${JSON.stringify(candidate)}\n`); return candidate; diff --git a/plugins/codexclaw/components/pabcd-state/dist/edit-shape.js b/plugins/codexclaw/components/pabcd-state/dist/edit-shape.js index 47055931..9f543e6a 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/edit-shape.js +++ b/plugins/codexclaw/components/pabcd-state/dist/edit-shape.js @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.js"; /** * edit-shape.ts — PostToolUse advisory for repeated same-shaped edits (astgrep_active 00). * @@ -20,7 +21,7 @@ * PostToolUse additionalContext envelope parity: omo lsp/src/codex-hook.ts:36-42. */ import { createHash } from "node:crypto"; -import { appendFileSync, mkdirSync, readFileSync } from "node:fs"; +import { appendFileSync, readFileSync } from "node:fs"; import { join } from "node:path"; import { splitLines } from "./text-lines.js"; @@ -163,7 +164,7 @@ function keyStates(rows ) { function appendRow(cwd , row ) { try { - mkdirSync(join(cwd, STATE_DIR), { recursive: true }); + ensureCodexclawDir(cwd); appendFileSync(ledgerPath(cwd), `${JSON.stringify(row)}\n`); } catch { // best-effort; advisory logic already ran on the in-memory state diff --git a/plugins/codexclaw/components/pabcd-state/dist/freeze-cli.js b/plugins/codexclaw/components/pabcd-state/dist/freeze-cli.js index bb8d5a33..e4ba1306 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/freeze-cli.js +++ b/plugins/codexclaw/components/pabcd-state/dist/freeze-cli.js @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.js"; /** * freeze-cli.ts — runtime wiring for the L10.3 freeze/stale path (HIGH-1/HIGH-4). * @@ -115,6 +116,7 @@ export function runFreeze(args ) { } if (!args.dryRun) { + ensureCodexclawDir(args.cwd); mkdirSync(join(args.cwd, STATE_DIR, FREEZE_MANIFEST_DIR), { recursive: true }); writeFileSync(manifestPath, JSON.stringify(manifest, null, 2)); } diff --git a/plugins/codexclaw/components/pabcd-state/dist/friction.js b/plugins/codexclaw/components/pabcd-state/dist/friction.js index f90dae53..e8cc7028 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/friction.js +++ b/plugins/codexclaw/components/pabcd-state/dist/friction.js @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.js"; /** * friction.ts — project-local friction ledger (lazygap_impl 080.1). * @@ -17,7 +18,7 @@ * (a read/parse error yields no verdict, so callers allow). */ import { createHash } from "node:crypto"; -import { appendFileSync, mkdirSync, readFileSync } from "node:fs"; +import { appendFileSync, readFileSync } from "node:fs"; import { join } from "node:path"; import { splitLines } from "./text-lines.js"; @@ -124,7 +125,7 @@ export function recordFriction(cwd , tool , errorText ) const verdict = verdictForCount(count); const entry = { ts: new Date().toISOString(), key, tool, normalized, count, verdict }; try { - mkdirSync(join(cwd, STATE_DIR), { recursive: true }); + ensureCodexclawDir(cwd); appendFileSync(frictionPath(cwd), `${JSON.stringify(entry)}\n`); } catch { // best-effort; the verdict is still meaningful to the caller diff --git a/plugins/codexclaw/components/pabcd-state/dist/goal-gate.js b/plugins/codexclaw/components/pabcd-state/dist/goal-gate.js index fc05fa9b..7dfc7c15 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/goal-gate.js +++ b/plugins/codexclaw/components/pabcd-state/dist/goal-gate.js @@ -39,6 +39,7 @@ import { captureSessionSourceIdentity } from "./session-source-identity.js"; import { resolveSessionSource } from "./session-source.js"; import { parseSourceBoundReceipt } from "./source-receipt.js"; import { hasSpentBudget, unrecordableVerdictStatus } from "./subagent-evidence.js"; +import { readPabcdEnabled } from "./interview-policy.js"; // Cross-component dist import (precedent: messenger-bridge/src/api-compat.ts:17). // 260724 WP1: deny remedies name `cxc orchestrate ...`/`cxc loop validate` — on a // payload-only install those must render the resolvable invocation. Emit-time only. @@ -206,7 +207,7 @@ function goalCompleteDenyEnvelope(reason ) { * Forensics: sessions 019f4407 (goal completed with a self-listed REMAINING queue) and * 019f4456 (empty goalplan would have rubber-stamped validate). */ -export function applyGoalCompleteGuard(payload ) { +export function applyGoalCompleteGuard(payload , pabcdEnabled = true) { try { if (payload.hook_event_name !== "PreToolUse") return ""; if (payload.tool_name !== UPDATE_GOAL_TOOL_NAME) return ""; @@ -223,7 +224,7 @@ export function applyGoalCompleteGuard(payload ) { `GOAL-COMPLETE-GATE-01: this session's state is unreadable, so unresolved subagent evidence failures cannot be ruled out. Restore or reset the session state after verifying the delegated work, or use update_goal status "blocked".`, ); } - if (state.orchestrationActive && state.phase !== "IDLE" && state.phase !== "I") { + if (pabcdEnabled && state.orchestrationActive && state.phase !== "IDLE" && state.phase !== "I") { return goalCompleteDenyEnvelope( `GOAL-COMPLETE-GATE-01: a PABCD cycle is in flight at phase ${state.phase}. Close the cycle first (advance to D via \`cxc orchestrate ... --session ${payload.session_id}\`, or \`cxc orchestrate reset --session ${payload.session_id}\`), then mark the goal complete. If an external blocker prevents closing, use update_goal status "blocked" instead.`, ); @@ -269,7 +270,7 @@ export function applyGoalCompleteGuard(payload ) { `GOAL-COMPLETE-GATE-01: a delegated subagent exhausted its evidence-verification budget without a valid receipt. Re-verify that work and record a receipt with \`cxc evidence resolve --session ${payload.session_id} --agent --receipt \`, or use update_goal status "blocked".`, ); } - if (state.slug) { + if (pabcdEnabled && state.slug) { try { resolveSessionSource(payload.cwd, payload.session_id); } catch (err) { return goalCompleteDenyEnvelope(`SOURCE-ROOT: ${err instanceof Error ? err.message : String(err)}`); } const plan = readGoalplan(payload.cwd, state.slug); @@ -314,9 +315,11 @@ export function handlePreToolUseFailClosed(raw , deps = { try { const payload = parsePreToolUse(raw); if (!payload) return ""; + const enabled = readPabcdEnabled(payload.cwd); // Each guard is tool-name-scoped, so at most one fires. - return applyGoalBudgetGuard(payload) || applyGoalModeInterviewGuard(payload, deps) || applyGoalCompleteGuard(payload); + return applyGoalBudgetGuard(payload) || (enabled ? applyGoalModeInterviewGuard(payload, deps) : "") || applyGoalCompleteGuard(payload, enabled); } catch { - return rawLooksLikeRequestUserInput(raw) ? goalModeInterviewDenyEnvelope("unreadable") : ""; + return readPabcdEnabled(process.cwd()) && rawLooksLikeRequestUserInput(raw) + ? goalModeInterviewDenyEnvelope("unreadable") : ""; } } diff --git a/plugins/codexclaw/components/pabcd-state/dist/goalplan.js b/plugins/codexclaw/components/pabcd-state/dist/goalplan.js index 381c9fc6..8d211227 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/goalplan.js +++ b/plugins/codexclaw/components/pabcd-state/dist/goalplan.js @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.js"; /** * goalplan.ts — project-local durable goalplan substrate (lazygap_impl 030). * @@ -850,6 +851,7 @@ function firstInvalidField(parsed ) { export function writeGoalplan(cwd , plan ) { validateGoalplanSlug(plan.slug); const dir = goalplanDir(cwd, plan.slug); + ensureCodexclawDir(cwd); mkdirSync(dir, { recursive: true, mode: 0o700 }); // Recheck after creation to close the ordinary pre-existing symlink case. goalplanDir(cwd, plan.slug); @@ -874,6 +876,7 @@ export function appendGoalplanLedger(cwd , slug , entry validateGoalplanSlug(slug); if (entry.slug !== slug) throw new Error("goalplan ledger entry slug does not match target slug"); const dir = goalplanDir(cwd, slug); + ensureCodexclawDir(cwd); mkdirSync(dir, { recursive: true, mode: 0o700 }); goalplanDir(cwd, slug); const path = goalplanLedgerPath(cwd, slug); diff --git a/plugins/codexclaw/components/pabcd-state/dist/hook.js b/plugins/codexclaw/components/pabcd-state/dist/hook.js index 602392c2..3320de19 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/hook.js +++ b/plugins/codexclaw/components/pabcd-state/dist/hook.js @@ -8,9 +8,9 @@ * * Stop: active under a native goal only. It returns a bounded * `{decision:"block",reason}` continuation envelope while a PABCD cycle is in flight — - * or, since 260709 (GOAL-IDLE-CONTINUE-01), while an ACTIVE goal is parked with no - * in-flight cycle (arming nudge). It releases on: no active goal, phase I, context - * pressure, or the same-phase stagnation cap (the single total-termination bound now + * or, since 260709 (GOAL-IDLE-CONTINUE-01), while an ACTIVE goal has a bound + * goalplan but no in-flight cycle (arming nudge). It releases on: no active goal, + * no bound plan, phase I, context pressure, or the same-phase stagnation cap (the single total-termination bound now * that the old unconditional `stop_hook_active` release is gone). * * Ground truth: @@ -26,6 +26,7 @@ import { matchesDcloseRecovery, readState, STATE_DIR, + statePath, writeState, @@ -230,22 +231,50 @@ export function resolveCxcInDirective(text ) { } } -/** - * Detect an explicit IPABCD/interview trigger. Explicit only — no goal-mode - * branch (A3 decision, see 022.3). Both English and Korean phrasings. - * Order matters: interview is checked first so "orchestrate i" wins over "p". - */ +/** Lines that can carry an advisory CodexClaw request, excluding quoted examples. */ +function requestLines(prompt ) { + const result = []; + let fenced = false; + for (const raw of (prompt ?? "").split(/\r?\n/)) { + const line = raw.trim(); + if (/^```/.test(line)) { fenced = !fenced; continue; } + if (fenced || !line || /^(?:>|[-*] |\d+[.)] )/.test(line)) continue; + // Explanatory leads describe a command; later mentions of docs/README do not. + const explanatory = /^(?:(?:please|좀)\s+)?(?:explain|describe|how do|how to|what is|what does|why)\b|^(?:좀\s*)?(?:설명|어떻게|뭐야)/i.test(line); + const unquoted = line + .replace(/`([^`]*)`/g, (match, inner , offset ) => { + if (explanatory || !/^(?:\$?(?:codexclaw:)?cxc-(?:loop|pabcd)|orchestrate\s+[ipabc])$/i.test(inner.trim())) return " "; + const before = line.slice(0, offset); + const after = line.slice(offset + match.length); + const addressed = /^(?:(?:please|좀)\s+)?(?:run|use|start|invoke|실행|돌려)\s*$/i.test(before) + || (/^(?:(?:please|좀)\s*)?$/i.test(before) && /^\s*(?:로|으로|써서)/.test(after)); + return addressed ? inner : " "; + }) + .replace(/"(?:\\.|[^"\\])*"|“[^”]*”|(? MAX_STOP_BLOCKS || nextTotal > MAX_STOP_BLOCKS_TOTAL) { - // give up the loop: reset the counter and release so the turn can end. - writeState(cwd, { ...state, ...carry, stopBlockPhase: null, stopBlockWorkPhaseId: null, stopBlockCount: 0 }); - return "release"; + const totalCap = nextTotal > MAX_STOP_BLOCKS_TOTAL; + const alreadyNotified = state.stopBlockCapNotified === true; + writeState(cwd, { ...state, ...carry, stopBlockPhase: null, stopBlockWorkPhaseId: null, + stopBlockCount: 0, stopBlockCapNotified: totalCap ? true : state.stopBlockCapNotified }); + if (!totalCap) return "phase-cap"; + return alreadyNotified ? "total-cap-silent" : "total-cap"; } writeState(cwd, { ...state, @@ -1643,14 +1665,12 @@ export function readStopWorkContext(cwd , state ) } /** - * GOAL-IDLE-CONTINUE-01 (260709) — the Stop block for "goal ACTIVE but no PABCD cycle - * in flight". The old guard 2a released this state silently, so a session could park an - * active goal at IDLE forever (019f4407: goal created, FSM never entered, turn ended). + * GOAL-IDLE-CONTINUE-01 (260709) — the Stop block for a goal ACTIVE with a bound + * goalplan but no PABCD cycle in flight. An unbound goal releases at IDLE. * The reason names the two honest exits: arm the next work-phase (`orchestrate P`), or * close the goal for real (`update_goal complete` — gated by GOAL-COMPLETE-GATE-01 when - * a goalplan is bound — or `blocked` for external blockers). When a goalplan is bound, - * the remaining work is named; when it is bound but unregistered (empty), the block says - * to fill it; when none is bound, it points at `cxc loop init`. + * a goalplan is bound — or `blocked` for external blockers). The remaining work is + * named when present; an empty bound plan gets guidance to register work phases. */ export function buildGoalIdleBlock( cwd , @@ -1746,9 +1766,10 @@ function objectivePlateau(cwd , sessionId ) { /** * Stop handler — L6 active continuation with a bounded stagnation guard so the loop * ALWAYS terminates. Blocks (keeps the agent going) only when a PABCD cycle is genuinely - * in flight under an active goal, OR when an ACTIVE goal is parked with no in-flight - * cycle (GOAL-IDLE-CONTINUE-01: arming nudge). Releases via any of: no active goal, - * phase I (interview firewall), context pressure, or the MAX_STOP_BLOCKS cap. + * in flight under an active goal, OR when an ACTIVE goal with a bound plan is + * parked at IDLE (GOAL-IDLE-CONTINUE-01: arming nudge). Releases via any of: + * no active goal, no bound plan at IDLE, phase I (interview firewall), context + * pressure, or the MAX_STOP_BLOCKS cap. * * 260709 (lazygap loop-enforcement patch): * - guard 1 (`stop_hook_active` → unconditional release) is REMOVED. Under the old @@ -1757,13 +1778,10 @@ function objectivePlateau(cwd , sessionId ) { * phase progress, which is the "step-by-step cut" the loop doctrine forbids. * Termination stays total: the per-phase MAX_STOP_BLOCKS stagnation cap (reset on * every real transition) bounds every continuation chain that stops progressing. - * - GOAL-IDLE-CONTINUE-01: an ACTIVE goal with no in-flight cycle used to release - * silently (guard 2a), so "goal armed but PABCD never entered" (019f4407) ended - * turns freely. It now gets the same bounded block, naming the arming command - * (`cxc orchestrate P --session `), the goalplan's remaining work when one is - * bound, and the honest close-out path (update_goal complete gated by E8 / blocked). - * Side effect by design: the counter write creates the session state file, so the - * suggested orchestrate command passes the G2 unknown-session guard afterwards. + * - GOAL-IDLE-CONTINUE-01: an ACTIVE goal with a resolvable bound plan gets a + * bounded block naming the arming command (`cxc orchestrate P --session `), + * remaining work, and the honest close-out path (update_goal complete gated by + * E8 / blocked). An unbound or stale slug releases without a counter write. */ export function handleStop( payload , @@ -1782,13 +1800,16 @@ export function handleStop( const inFlight = state.orchestrationActive && state.phase !== "IDLE"; // guard 2a (amended by GOAL-IDLE-CONTINUE-01): with no cycle in flight a plain - // interactive session releases exactly as before; an ACTIVE goal instead gets a - // bounded arming block — "IDLE is not the end while work remains" (LOOP-CONTINUE-01). + // interactive session releases exactly as before; an ACTIVE goal with a bound + // plan gets a bounded arming block — "IDLE is not the end while work remains". if (!inFlight) { if (!goalActive) return ""; + if (!state.slug || !safeReadBoundGoalplan(payload.cwd, state.slug)) return ""; // bail: don't pile on during context-pressure/compaction recovery. if (isContextPressureTail(readTranscriptTail(payload.transcript_path))) return ""; - if (bumpStopCounter(payload.cwd, state) === "release") return ""; + const count = bumpStopCounter(payload.cwd, state); + if (count === "total-cap") return `${JSON.stringify({ systemMessage: "CodexClaw Stop continuation cap (24) reached for this user turn; releasing." })}\n`; + if (typeof count !== "number") return ""; return buildGoalIdleBlock(payload.cwd, state, payload.session_id, platform); } @@ -1807,7 +1828,9 @@ export function handleStop( // bail: don't pile on during context-pressure/compaction recovery. if (isContextPressureTail(readTranscriptTail(payload.transcript_path))) return ""; - if (bumpStopCounter(payload.cwd, state) === "release") return ""; + const count = bumpStopCounter(payload.cwd, state); + if (count === "total-cap") return `${JSON.stringify({ systemMessage: "CodexClaw Stop continuation cap (24) reached for this user turn; releasing." })}\n`; + if (typeof count !== "number") return ""; const plateau = objectivePlateau(payload.cwd, payload.session_id); if (plateau.flat) return buildPlateauDivergeBlock(state.phase, plateau, payload.cwd, payload.session_id); // 040: enrich the block reason with goalplan-derived remaining work (text-only, after diff --git a/plugins/codexclaw/components/pabcd-state/dist/interview-ledger.js b/plugins/codexclaw/components/pabcd-state/dist/interview-ledger.js index 160bd616..cf296c72 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/interview-ledger.js +++ b/plugins/codexclaw/components/pabcd-state/dist/interview-ledger.js @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.js"; /** * interview-ledger.ts — durable interview question/answer capture (L12 / 120, WP4). * @@ -217,6 +218,7 @@ function alreadyRecorded(cwd , sessionId , eventId ) function appendEvent(cwd , entry ) { const dir = join(cwd, STATE_DIR, INTERVIEWS_SUBDIR); + ensureCodexclawDir(cwd); mkdirSync(dir, { recursive: true }); appendFileSync(ledgerPath(cwd, entry.sessionId), `${JSON.stringify(entry)}\n`); } diff --git a/plugins/codexclaw/components/pabcd-state/dist/interview-policy.js b/plugins/codexclaw/components/pabcd-state/dist/interview-policy.js index 8974a162..6b685309 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/interview-policy.js +++ b/plugins/codexclaw/components/pabcd-state/dist/interview-policy.js @@ -43,6 +43,20 @@ export function configPath(cwd ) { return join(cwd, CONFIG_FILENAME); } +/** PABCD hook policy: a recognized environment value overrides project config. */ +export function readPabcdEnabled(cwd , env = process.env) { + const override = env.CODEXCLAW_PABCD?.trim().toLowerCase(); + if (override === "off" || override === "0" || override === "false") return false; + if (override === "on" || override === "1" || override === "true") return true; + try { + const raw = JSON.parse(readFileSync(configPath(cwd), "utf8")); + if (!raw || typeof raw !== "object" || Array.isArray(raw)) return true; + const pabcd = (raw ).pabcd; + if (!pabcd || typeof pabcd !== "object" || Array.isArray(pabcd)) return true; + return (pabcd ).enabled !== false; + } catch { return true; } +} + /** * Read the policy for this repo. Missing file, unreadable file, malformed JSON and * unknown values all fall back to the default: a hook must never throw on a prompt. diff --git a/plugins/codexclaw/components/pabcd-state/dist/metrics.js b/plugins/codexclaw/components/pabcd-state/dist/metrics.js index 8e69931b..3807996d 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/metrics.js +++ b/plugins/codexclaw/components/pabcd-state/dist/metrics.js @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.js"; import { appendFileSync, mkdirSync, readFileSync, rmSync, writeFileSync } from "node:fs"; import { join } from "node:path"; import { renameWithRetry } from "./atomic-write.js"; @@ -129,12 +130,13 @@ export function recordObjectiveMetric(cwd , input best: Math.max(previousBest, input.value), source: input.source, }; - mkdirSync(codexclawDir(cwd), { recursive: true }); + ensureCodexclawDir(cwd); appendFileSync(metricsPath(cwd), `${JSON.stringify(next)}\n`); return next; } export function writeObjectiveKind(cwd , sessionId , kind ) { + ensureCodexclawDir(cwd); mkdirSync(objectiveKindDir(cwd), { recursive: true }); const finalPath = objectiveKindPath(cwd, sessionId); const tmp = `${finalPath}.${process.pid}.${Date.now()}.tmp`; diff --git a/plugins/codexclaw/components/pabcd-state/dist/receipt-cli.js b/plugins/codexclaw/components/pabcd-state/dist/receipt-cli.js index fdd52c44..a458f81c 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/receipt-cli.js +++ b/plugins/codexclaw/components/pabcd-state/dist/receipt-cli.js @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.js"; /** * receipt-cli.ts — `cxc receipt test` (075). * @@ -178,6 +179,7 @@ export function runReceiptCli(args ) { // Recorded so a reader can see WHICH rewrites the check was allowed to make. ...(args.generated && args.generated.length > 0 ? { generatedPaths: args.generated } : {}), }; + ensureCodexclawDir(args.cwd); mkdirSync(join(path, ".."), { recursive: true }); writeFileSync(path, `${JSON.stringify(receipt, null, 2)}\n`); return { output: path, code: 0 }; diff --git a/plugins/codexclaw/components/pabcd-state/dist/release-cli.js b/plugins/codexclaw/components/pabcd-state/dist/release-cli.js index 693f8744..b1eaa3f0 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/release-cli.js +++ b/plugins/codexclaw/components/pabcd-state/dist/release-cli.js @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.js"; /** * release-cli.ts — `cxc release`: assemble a candidate manifest from real receipts * and refuse publication when the evidence does not describe the candidate commit. @@ -131,7 +132,8 @@ function readCandidate(path ) } /** Atomic write so a crashed step never leaves a half-written candidate. */ -function writeCandidate(path , manifest ) { +function writeCandidate(path , manifest , projectCwd ) { + if (projectCwd) ensureCodexclawDir(projectCwd); mkdirSync(dirname(path), { recursive: true }); const tmp = path + ".tmp"; writeFileSync(tmp, JSON.stringify(manifest, null, 2) + "\n"); @@ -177,7 +179,7 @@ function runInit(argv , cwd ) { scorecard: {}, nonGoals: [], }; - writeCandidate(path, manifest); + writeCandidate(path, manifest, explicit ? undefined : cwd); return { code: 0, output: @@ -196,7 +198,7 @@ function mutate( if ("error" in read) return { code: 1, output: "release: " + read.error }; const outcome = fn(read.manifest); if (typeof outcome !== "string") return outcome; - writeCandidate(resolved.path, read.manifest); + writeCandidate(resolved.path, read.manifest, argv.includes("--candidate") ? undefined : cwd); return { code: 0, output: outcome }; } @@ -329,7 +331,7 @@ function runVerify(argv , cwd ) { // skipped, rather than leaving it in a CI log nobody reads. if (allowDeferred && manifest.allowedDeferred !== true) { manifest.allowedDeferred = true; - writeCandidate(resolved.path, manifest); + writeCandidate(resolved.path, manifest, argv.includes("--candidate") ? undefined : cwd); } if (asJson) { diff --git a/plugins/codexclaw/components/pabcd-state/dist/render-observations.js b/plugins/codexclaw/components/pabcd-state/dist/render-observations.js index 6e21fe21..df21611c 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/render-observations.js +++ b/plugins/codexclaw/components/pabcd-state/dist/render-observations.js @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.js"; /** * render-observations.ts — L2 render-observation ledger (C-RENDER-GROUNDING-01). * @@ -18,7 +19,7 @@ * All IO is project-local under `cwd`. Every reader FAILS-OPEN (missing file or * parse error yields []). */ -import { appendFileSync, mkdirSync, readFileSync, readdirSync, statSync, writeFileSync } from "node:fs"; +import { appendFileSync, readFileSync, readdirSync, statSync, writeFileSync } from "node:fs"; import { dirname, extname, join, resolve } from "node:path"; import { fileEditShapes } from "./edit-shape.js"; @@ -69,7 +70,7 @@ function ledgerPath(cwd ) { function appendRow(cwd , row ) { try { - mkdirSync(join(cwd, STATE_DIR), { recursive: true }); + ensureCodexclawDir(cwd); appendFileSync(ledgerPath(cwd), `${JSON.stringify(row)}\n`); } catch { // best-effort; advisory logic runs on in-memory state @@ -124,7 +125,7 @@ export function readRenderObsRows(cwd ) { */ export function resetRenderLedger(cwd ) { try { - mkdirSync(join(cwd, STATE_DIR), { recursive: true }); + ensureCodexclawDir(cwd); writeFileSync(ledgerPath(cwd), ""); } catch { // best-effort diff --git a/plugins/codexclaw/components/pabcd-state/dist/rule-impact-ledger.js b/plugins/codexclaw/components/pabcd-state/dist/rule-impact-ledger.js index b8ec4330..4f8cb3f2 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/rule-impact-ledger.js +++ b/plugins/codexclaw/components/pabcd-state/dist/rule-impact-ledger.js @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.js"; /** * rule-impact-ledger.ts — opt-in Rule Impact Ledger (issue #18). * @@ -5,7 +6,7 @@ * outcome change, and cost so the router diet can be evidence-based. */ import { writeFileSync, readFileSync, existsSync, mkdirSync } from "node:fs"; -import { join, dirname } from "node:path"; +import { join, dirname, resolve, sep } from "node:path"; /** Schema version for the rule impact ledger. */ export const LEDGER_SCHEMA_VERSION = 1; @@ -102,6 +103,10 @@ export function appendRecord(path , record ) { const ledger = readLedger(path); ledger.records.push(record); const dir = dirname(path); + const absolute = resolve(path); + const marker = `${sep}.codexclaw${sep}`; + const index = absolute.indexOf(marker); + if (index >= 0) ensureCodexclawDir(absolute.slice(0, index)); if (!existsSync(dir)) mkdirSync(dir, { recursive: true }); writeFileSync(path, JSON.stringify(ledger, null, 2)); } diff --git a/plugins/codexclaw/components/pabcd-state/dist/session-source.js b/plugins/codexclaw/components/pabcd-state/dist/session-source.js index c7da8bca..55c0ec96 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/session-source.js +++ b/plugins/codexclaw/components/pabcd-state/dist/session-source.js @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.js"; /** Immutable per-session source binding; native state/identity never moves. */ import { execFileSync } from "node:child_process"; import { randomUUID } from "node:crypto"; @@ -180,6 +181,7 @@ export function bindSessionSource(cwd , sessionId , target } const binding = { version: 1, ownerSessionId: sessionId, nativeCwd, sourceRoot, commonDir: source.commonDir, gitDir: source.gitDir }; const path = bindingPath(cwd, sessionId); + ensureCodexclawDir(cwd); mkdirSync(join(cwd, ".codexclaw", "sources"), { recursive: true }); bindingPath(cwd, sessionId); const tmp = `${path}.${randomUUID()}.tmp`; diff --git a/plugins/codexclaw/components/pabcd-state/dist/state.js b/plugins/codexclaw/components/pabcd-state/dist/state.js index 01b0c226..82512719 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/state.js +++ b/plugins/codexclaw/components/pabcd-state/dist/state.js @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.js"; import { existsSync, mkdirSync, readFileSync, writeFileSync, appendFileSync, linkSync, rmSync, statSync } from "node:fs"; import { randomUUID } from "node:crypto"; import { isAbsolute, join, resolve } from "node:path"; @@ -225,6 +226,10 @@ export function reconstructUnverified(raw ) + + + + @@ -298,6 +303,8 @@ export function defaultState(sessionId , slug = "") { stopBlockWorkPhaseId: null, stopMetricCursor: 0, stopBlockTotal: 0, + stopBlockTurnId: null, + stopBlockCapNotified: false, loopArmSeen: false, idleEditNudges: 0, memoryWriteRequested: false, @@ -317,7 +324,7 @@ function sessionsDir(cwd ) { return join(cwd, STATE_DIR, SESSIONS_SUBDIR); } -function statePath(cwd , sessionId ) { +export function statePath(cwd , sessionId ) { return join(sessionsDir(cwd), `${sanitizeKey(sessionId)}.json`); } @@ -374,6 +381,7 @@ export function ensureState( throw new TypeError("sessionId must be a canonical state key"); } const dir = sessionsDir(cwd); + ensureCodexclawDir(cwd); mkdirSync(dir, { recursive: true }); const finalPath = statePath(cwd, sessionId); const tmp = `${finalPath}.${process.pid}.${randomUUID()}.tmp`; @@ -552,6 +560,11 @@ export function readStateStrict(cwd , sessionId ) typeof parsed.stopBlockTotal === "number" && Number.isFinite(parsed.stopBlockTotal) && parsed.stopBlockTotal >= 0 ? Math.floor(parsed.stopBlockTotal) : 0, + stopBlockTurnId: + typeof parsed.stopBlockTurnId === "string" && parsed.stopBlockTurnId.length > 0 + ? parsed.stopBlockTurnId + : null, + stopBlockCapNotified: parsed.stopBlockCapNotified === true, // 260714 wp3: strict reconstruction (old files read false/0 — backward-compatible). loopArmSeen: parsed.loopArmSeen === true, idleEditNudges: @@ -606,6 +619,7 @@ export function readStateStrict(cwd , sessionId ) export function writeState(cwd , next ) { const dir = sessionsDir(cwd); + ensureCodexclawDir(cwd); mkdirSync(dir, { recursive: true }); const finalPath = statePath(cwd, next.sessionId); const tmp = `${finalPath}.${process.pid}.${Date.now()}.tmp`; @@ -653,6 +667,7 @@ function sleepSyncMs(ms ) { export function withSessionLock (cwd , sessionId , fn ) { const dir = sessionsDir(cwd); + ensureCodexclawDir(cwd); mkdirSync(dir, { recursive: true }); const lockPath = `${statePath(cwd, sessionId)}.lock`; let held = false; @@ -680,6 +695,7 @@ export function withSessionLock (cwd , sessionId , fn ) export function appendLedger(cwd , entry ) { const dir = join(cwd, STATE_DIR); + ensureCodexclawDir(cwd); mkdirSync(dir, { recursive: true }); appendFileSync(join(dir, LEDGER_FILE), `${JSON.stringify(entry)}\n`); } @@ -737,6 +753,7 @@ function interviewLedgerPath(cwd , sessionId ) { */ export function appendInterviewEvent(cwd , entry ) { const dir = interviewsDir(cwd); + ensureCodexclawDir(cwd); mkdirSync(dir, { recursive: true }); appendFileSync(interviewLedgerPath(cwd, entry.sessionId), `${JSON.stringify(entry)}\n`); } diff --git a/plugins/codexclaw/components/pabcd-state/dist/subagent-evidence.js b/plugins/codexclaw/components/pabcd-state/dist/subagent-evidence.js index 6c6eb68e..24fb0bd3 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/subagent-evidence.js +++ b/plugins/codexclaw/components/pabcd-state/dist/subagent-evidence.js @@ -1,7 +1,8 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.js"; /** * subagent-evidence.ts — SubagentStop evidence-receipt gate (lazygap_impl 010). * - * A dispatched WRITE/verify subagent (agent_type "executor", or legacy "worker") cannot "finish" without a + * A registered executor, or a legacy worker in an active PABCD B/C cycle, cannot "finish" without a * non-empty evidence receipt under `.codexclaw/evidence/`. Missing/invalid receipt -> * `decision:"block"` with a verifier directive that re-prompts the CHILD (codex-rs * turn.rs:323). After MAX_ATTEMPTS the directive escalates but remains fail-closed; @@ -28,6 +29,7 @@ * (transcript_path = parent, agent_transcript_path = child): schema.rs:576, hook_runtime.rs:302 * - decision:"block" + reason re-prompts the child's own turn: stop.rs:263,351 + turn.rs:323 */ +import { readPabcdEnabled } from "./interview-policy.js"; import { existsSync, lstatSync, @@ -55,8 +57,8 @@ import { /** - * agent_type values this gate refuses to release without a receipt. - * DISPATCH-AGENT-TYPE-01: executor and legacy worker are gated. Read-only audit/research + * agent_type values routed to this gate. + * DISPATCH-AGENT-TYPE-01: executor and legacy worker are candidates. Read-only audit/research * dispatches MUST use agent_type:"explorer" so they bypass both the hook * manifest matcher (^(executor|worker)$) and this runtime gate. See * structure/20_pabcd_dispatch_doctrine.md §3. @@ -214,6 +216,7 @@ export function readAttempts(cwd , sessionId , agentId , tu export function writeAttempts(cwd , sessionId , agentId , attempts , turnId = "") { try { const p = attemptsPath(cwd, sessionId, agentId, turnId); + ensureCodexclawDir(cwd); mkdirSync(join(cwd, STATE_DIR, EVIDENCE_ATTEMPTS_SUBDIR), { recursive: true }); const tmp = `${p}.${process.pid}.${Date.now()}.tmp`; writeFileSync(tmp, `${JSON.stringify({ attempts })}\n`); @@ -335,6 +338,7 @@ function unrecordableDir(cwd ) { */ export function writeUnrecordableMarker(cwd , sessionId , agentId ) { const dir = unrecordableDir(cwd); + ensureCodexclawDir(cwd); mkdirSync(dir, { recursive: true }); const p = join(dir, `${sanitizeKey(sessionId)}-${sanitizeKey(agentId)}-${Date.now()}.json`); writeFileSync(p, `${JSON.stringify({ sessionId, agentId, at: new Date().toISOString() })}\n`, { flag: "wx" }); @@ -350,6 +354,7 @@ export function writeUnrecordableMarker(cwd , sessionId , agentId function markerDirWritable(cwd ) { const probe = join(unrecordableDir(cwd), `.probe-${process.pid}-${Date.now()}`); try { + ensureCodexclawDir(cwd); mkdirSync(unrecordableDir(cwd), { recursive: true }); writeFileSync(probe, "", { flag: "wx" }); rmSync(probe, { force: true }); @@ -470,7 +475,12 @@ export function escalationDirective() { */ export function runSubagentStopGate(payload ) { try { + if (!readPabcdEnabled(payload.cwd)) return ""; if (!GATED_AGENT_TYPES.has(payload.agent_type)) return ""; + if (payload.agent_type === "worker") { + const { state, unreadable } = readStateStrict(payload.cwd, payload.session_id); + if (unreadable || !state.orchestrationActive || (state.phase !== "B" && state.phase !== "C")) return ""; + } const agentId = payload.agent_id ?? ""; const { cwd, session_id: sessionId } = payload; // Same identity as a tombstone: two turns of one agent must not share a budget. diff --git a/plugins/codexclaw/components/pabcd-state/dist/worktree-guard.js b/plugins/codexclaw/components/pabcd-state/dist/worktree-guard.js index 9940c842..27b622d5 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/worktree-guard.js +++ b/plugins/codexclaw/components/pabcd-state/dist/worktree-guard.js @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.js"; /** * worktree-guard.ts — Codex-app managed-worktree identity guard (260804 unit, * devlog/_plan/260804_worktree_identity_guardian/010 rev2). @@ -524,6 +525,7 @@ function alreadyInjected(cwd , sessionId ) { function markInjected(cwd , sessionId , slot ) { try { const path = markerPath(cwd, sessionId); + ensureCodexclawDir(cwd); mkdirSync(resolve(path, ".."), { recursive: true }); writeFileSync(path, JSON.stringify({ injectedAt: new Date().toISOString(), slot }), { encoding: "utf8", diff --git a/plugins/codexclaw/components/subagent-config/dist/fallback-dispatch.js b/plugins/codexclaw/components/subagent-config/dist/fallback-dispatch.js index 8659668d..ad06a057 100644 --- a/plugins/codexclaw/components/subagent-config/dist/fallback-dispatch.js +++ b/plugins/codexclaw/components/subagent-config/dist/fallback-dispatch.js @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.js"; /** Main-owned native spawn protocol. This module selects attempts; it never calls a provider. */ import { randomUUID } from "node:crypto"; import { existsSync, lstatSync, mkdirSync, readFileSync, realpathSync, rmSync, writeFileSync } from "node:fs"; @@ -68,8 +69,9 @@ function taskFailure(raw ) { return { kind: t.kind , evidence: smallText(t.evidence, "taskFailure evidence") }; } function directory(cwd , sessionId ) { - let dir = cwd; - for (const part of [".codexclaw", "dispatches", sessionId]) { + ensureCodexclawDir(cwd); + let dir = join(cwd, ".codexclaw"); + for (const part of ["dispatches", sessionId]) { dir = join(dir, part); if (existsSync(dir)) { if (!lstatSync(dir).isDirectory() || lstatSync(dir).isSymbolicLink()) throw new Error("dispatch directory must not be a symlink"); diff --git a/plugins/codexclaw/components/subagent-config/dist/spawn-wrapper.js b/plugins/codexclaw/components/subagent-config/dist/spawn-wrapper.js index 8a1b485c..4983d3bd 100644 --- a/plugins/codexclaw/components/subagent-config/dist/spawn-wrapper.js +++ b/plugins/codexclaw/components/subagent-config/dist/spawn-wrapper.js @@ -10,6 +10,8 @@ * explorer/reviewer -> "explorer", executor -> "worker". Architect never aliases another role. * executor resolves to its registered native "executor" type when $CODEX_HOME/agents/executor.toml * exists (cxc subagents register executor); unregistered installs keep built-in worker. + * With PABCD enabled, executor is receipt-gated on every stop; worker is gated only + * while the parent is actively orchestrating B/C. Both release when policy is off. * - the role prompt is injected INLINE in the message ("TASK: ..."), since plugin * install dirs are not a config layer. * - model selection is not emitted by the v2 builder. The durable per-role model in diff --git a/plugins/codexclaw/components/subagent-config/dist/store.js b/plugins/codexclaw/components/subagent-config/dist/store.js index 9fe12e52..86a19ec1 100644 --- a/plugins/codexclaw/components/subagent-config/dist/store.js +++ b/plugins/codexclaw/components/subagent-config/dist/store.js @@ -1,3 +1,4 @@ +import { ensureCodexclawDir } from "./codexclaw-dir.js"; /** * store.ts — `.codexclaw/subagents.json` config store (L24 / 240-242). * @@ -249,7 +250,8 @@ export function validateRolePatch(patch ) { } /** Atomic write with an exclusive temporary file; preserve unrelated JSON fields. */ -function writeRaw(path , config ) { +function writeRaw(path , config , projectCwd ) { + if (projectCwd) ensureCodexclawDir(projectCwd); mkdirSync(dirname(path), { recursive: true, mode: 0o700 }); const tmp = `${path}.${randomUUID()}.tmp`; try { @@ -262,7 +264,7 @@ function writeRaw(path , config ) { /** Explicit full-config writes remain available to existing callers. */ export function writeConfig(cwd , config ) { - writeRaw(storePath(cwd), config); + writeRaw(storePath(cwd), config, cwd); } /** Merge only the selected role; missing roles continue to inherit dynamically. */ @@ -283,7 +285,7 @@ export function setRole(cwd , role , patch , scope if (next.mode === "default") next.model = null; if (next.fallback) next.fallback = { model: next.fallback.model, effort: next.fallback.effort }; raw.roles[role] = { ...(typeof raw.roles[role] === "object" && raw.roles[role] !== null ? raw.roles[role] : {}), ...next }; - writeRaw(path, raw); + writeRaw(path, raw, scope === "project" ? cwd : undefined); return readConfig(cwd, scope, env); } @@ -294,7 +296,7 @@ export function resetRole(cwd , role , scope = "pro const raw = scope === "global" ? readGlobalRaw(env, true) : readRaw(path, true); if (Object.hasOwn(raw.roles, role)) { delete raw.roles[role]; - writeRaw(path, raw); + writeRaw(path, raw, scope === "project" ? cwd : undefined); } return readConfig(cwd, scope, env); } From 216e15b3b95c83be7e463937274d7c10bf20df31 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 01:04:53 +0900 Subject: [PATCH 25/90] build: track new codexclaw-dir dist outputs --- .../components/bg-wake/dist/codexclaw-dir.js | 26 +++++++++++++++++++ .../components/cxc-ops/dist/codexclaw-dir.js | 26 +++++++++++++++++++ .../messenger-bridge/dist/codexclaw-dir.js | 26 +++++++++++++++++++ .../pabcd-state/dist/codexclaw-dir.js | 26 +++++++++++++++++++ .../subagent-config/dist/codexclaw-dir.js | 26 +++++++++++++++++++ 5 files changed, 130 insertions(+) create mode 100644 plugins/codexclaw/components/bg-wake/dist/codexclaw-dir.js create mode 100644 plugins/codexclaw/components/cxc-ops/dist/codexclaw-dir.js create mode 100644 plugins/codexclaw/components/messenger-bridge/dist/codexclaw-dir.js create mode 100644 plugins/codexclaw/components/pabcd-state/dist/codexclaw-dir.js create mode 100644 plugins/codexclaw/components/subagent-config/dist/codexclaw-dir.js diff --git a/plugins/codexclaw/components/bg-wake/dist/codexclaw-dir.js b/plugins/codexclaw/components/bg-wake/dist/codexclaw-dir.js new file mode 100644 index 00000000..d059445d --- /dev/null +++ b/plugins/codexclaw/components/bg-wake/dist/codexclaw-dir.js @@ -0,0 +1,26 @@ +import { mkdirSync, rmdirSync, writeFileSync } from "node:fs"; +import { join } from "node:path"; + +export const GITIGNORE_TEXT = "# CodexClaw wrote this when it created .codexclaw; everything here is local runtime state.\n*\n!.gitignore\n!rules/\n!rules/*.md\n"; + +/** Only the process that first creates the project-local root may publish its ignore file. */ +export function ensureCodexclawDir( + cwd , + writeIgnore = writeFileSync, +) { + const dir = join(cwd, ".codexclaw"); + try { + mkdirSync(dir); + } catch (err) { + if ((err )?.code === "EEXIST") return dir; + throw err; + } + try { + writeIgnore(join(dir, ".gitignore"), GITIGNORE_TEXT, { flag: "wx" }); + } catch (err) { + if ((err )?.code === "EEXIST") return dir; + try { rmdirSync(dir); } catch { /* A concurrent writer made the directory non-empty. */ } + throw err; + } + return dir; +} diff --git a/plugins/codexclaw/components/cxc-ops/dist/codexclaw-dir.js b/plugins/codexclaw/components/cxc-ops/dist/codexclaw-dir.js new file mode 100644 index 00000000..d059445d --- /dev/null +++ b/plugins/codexclaw/components/cxc-ops/dist/codexclaw-dir.js @@ -0,0 +1,26 @@ +import { mkdirSync, rmdirSync, writeFileSync } from "node:fs"; +import { join } from "node:path"; + +export const GITIGNORE_TEXT = "# CodexClaw wrote this when it created .codexclaw; everything here is local runtime state.\n*\n!.gitignore\n!rules/\n!rules/*.md\n"; + +/** Only the process that first creates the project-local root may publish its ignore file. */ +export function ensureCodexclawDir( + cwd , + writeIgnore = writeFileSync, +) { + const dir = join(cwd, ".codexclaw"); + try { + mkdirSync(dir); + } catch (err) { + if ((err )?.code === "EEXIST") return dir; + throw err; + } + try { + writeIgnore(join(dir, ".gitignore"), GITIGNORE_TEXT, { flag: "wx" }); + } catch (err) { + if ((err )?.code === "EEXIST") return dir; + try { rmdirSync(dir); } catch { /* A concurrent writer made the directory non-empty. */ } + throw err; + } + return dir; +} diff --git a/plugins/codexclaw/components/messenger-bridge/dist/codexclaw-dir.js b/plugins/codexclaw/components/messenger-bridge/dist/codexclaw-dir.js new file mode 100644 index 00000000..d059445d --- /dev/null +++ b/plugins/codexclaw/components/messenger-bridge/dist/codexclaw-dir.js @@ -0,0 +1,26 @@ +import { mkdirSync, rmdirSync, writeFileSync } from "node:fs"; +import { join } from "node:path"; + +export const GITIGNORE_TEXT = "# CodexClaw wrote this when it created .codexclaw; everything here is local runtime state.\n*\n!.gitignore\n!rules/\n!rules/*.md\n"; + +/** Only the process that first creates the project-local root may publish its ignore file. */ +export function ensureCodexclawDir( + cwd , + writeIgnore = writeFileSync, +) { + const dir = join(cwd, ".codexclaw"); + try { + mkdirSync(dir); + } catch (err) { + if ((err )?.code === "EEXIST") return dir; + throw err; + } + try { + writeIgnore(join(dir, ".gitignore"), GITIGNORE_TEXT, { flag: "wx" }); + } catch (err) { + if ((err )?.code === "EEXIST") return dir; + try { rmdirSync(dir); } catch { /* A concurrent writer made the directory non-empty. */ } + throw err; + } + return dir; +} diff --git a/plugins/codexclaw/components/pabcd-state/dist/codexclaw-dir.js b/plugins/codexclaw/components/pabcd-state/dist/codexclaw-dir.js new file mode 100644 index 00000000..d059445d --- /dev/null +++ b/plugins/codexclaw/components/pabcd-state/dist/codexclaw-dir.js @@ -0,0 +1,26 @@ +import { mkdirSync, rmdirSync, writeFileSync } from "node:fs"; +import { join } from "node:path"; + +export const GITIGNORE_TEXT = "# CodexClaw wrote this when it created .codexclaw; everything here is local runtime state.\n*\n!.gitignore\n!rules/\n!rules/*.md\n"; + +/** Only the process that first creates the project-local root may publish its ignore file. */ +export function ensureCodexclawDir( + cwd , + writeIgnore = writeFileSync, +) { + const dir = join(cwd, ".codexclaw"); + try { + mkdirSync(dir); + } catch (err) { + if ((err )?.code === "EEXIST") return dir; + throw err; + } + try { + writeIgnore(join(dir, ".gitignore"), GITIGNORE_TEXT, { flag: "wx" }); + } catch (err) { + if ((err )?.code === "EEXIST") return dir; + try { rmdirSync(dir); } catch { /* A concurrent writer made the directory non-empty. */ } + throw err; + } + return dir; +} diff --git a/plugins/codexclaw/components/subagent-config/dist/codexclaw-dir.js b/plugins/codexclaw/components/subagent-config/dist/codexclaw-dir.js new file mode 100644 index 00000000..d059445d --- /dev/null +++ b/plugins/codexclaw/components/subagent-config/dist/codexclaw-dir.js @@ -0,0 +1,26 @@ +import { mkdirSync, rmdirSync, writeFileSync } from "node:fs"; +import { join } from "node:path"; + +export const GITIGNORE_TEXT = "# CodexClaw wrote this when it created .codexclaw; everything here is local runtime state.\n*\n!.gitignore\n!rules/\n!rules/*.md\n"; + +/** Only the process that first creates the project-local root may publish its ignore file. */ +export function ensureCodexclawDir( + cwd , + writeIgnore = writeFileSync, +) { + const dir = join(cwd, ".codexclaw"); + try { + mkdirSync(dir); + } catch (err) { + if ((err )?.code === "EEXIST") return dir; + throw err; + } + try { + writeIgnore(join(dir, ".gitignore"), GITIGNORE_TEXT, { flag: "wx" }); + } catch (err) { + if ((err )?.code === "EEXIST") return dir; + try { rmdirSync(dir); } catch { /* A concurrent writer made the directory non-empty. */ } + throw err; + } + return dir; +} From 5fcf859c65581b0c768d105959a4e5c86785f64e Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 01:05:24 +0900 Subject: [PATCH 26/90] docs: publish measured test count (3649) --- README.ko.md | 2 +- README.md | 2 +- README.zh.md | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/README.ko.md b/README.ko.md index 5f43b04a..7bd91054 100644 --- a/README.ko.md +++ b/README.ko.md @@ -13,7 +13,7 @@

CI - 3,609 tests + 3,649 tests 29 skills 29 hooks Documentation diff --git a/README.md b/README.md index 20447a2f..428fa714 100644 --- a/README.md +++ b/README.md @@ -13,7 +13,7 @@

CI - 3,609 tests + 3,649 tests 29 skills 29 hooks Documentation diff --git a/README.zh.md b/README.zh.md index 26786fed..9975d663 100644 --- a/README.zh.md +++ b/README.zh.md @@ -13,7 +13,7 @@

CI - 3,609 tests + 3,649 tests 29 skills 29 hooks Documentation From a9f6826d986a72012c3ea675151afbe196686cff Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 01:05:34 +0900 Subject: [PATCH 27/90] docs(changelog): wp2 hook runtime fixes --- CHANGELOG.md | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 896e7677..117e214a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,22 @@ All notable changes to codexclaw are documented here. The format follows ## [Unreleased] +### Added + +- `CODEXCLAW_PABCD=off` (or `on`) and project `codexclaw.json` `{"pabcd": {"enabled": false}}` turn the PABCD hook policy off while keeping the worktree, memory-write, automation-ownership and apply_patch lint guards and recall active. A recognized environment value wins over the project file in both directions (#252). +- When codexclaw creates a project's `.codexclaw` folder, it also writes `.codexclaw/.gitignore` so session state, ledgers and evidence stay out of git; user-authored `rules/*.md` stay committable unless an ancestor ignore rule hides the folder. Existing `.codexclaw` folders are never modified. Lazy creation of session state is deferred (#255, partial). + +### Changed + +- The absolute Stop continuation cap (24) now counts per genuine user turn instead of per session, and the release prints one notice per turn (#254). + +### Fixed + +- Ordinary words (for example "interview", "keep going until", "끝까지 진행해", quoted or fenced examples) no longer inject PABCD phase directives or arm the loop; hints need an explicit codexclaw request such as `cxc-pabcd` or `cxc-loop` (#250). +- The SubagentStop evidence gate no longer blocks Codex's built-in `worker` outside an active PABCD build or check cycle; registered `executor` stays gated while PABCD is on (#251). +- An active native goal without a bound goalplan no longer blocks Stop at IDLE (#253). + + ## [0.2.39] - 2026-09-24 ### Added From 13bac9d6626b83901b9e5eee12db16d91f2473c8 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 01:11:24 +0900 Subject: [PATCH 28/90] fix(pabcd-state): negated requests never arm or inject (#250) --- .../components/pabcd-state/dist/hook.js | 11 +++++++- .../components/pabcd-state/src/hook.ts | 11 +++++++- .../components/pabcd-state/test/hook.test.ts | 25 +++++++++++++++++++ 3 files changed, 45 insertions(+), 2 deletions(-) diff --git a/plugins/codexclaw/components/pabcd-state/dist/hook.js b/plugins/codexclaw/components/pabcd-state/dist/hook.js index 3320de19..3aef912c 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/hook.js +++ b/plugins/codexclaw/components/pabcd-state/dist/hook.js @@ -252,11 +252,20 @@ function requestLines(prompt ) { }) .replace(/"(?:\\.|[^"\\])*"|“[^”]*”|(? { ] as const) assert.equal(detectTrigger(prompt), phase, prompt); }); + +test("issue 250: negated requests never arm or inject", () => { + for (const prompt of [ + "Do not run cxc-loop for this task", + "Do not use cxc-pabcd to start Plan phase", + "cxc-loop 돌리지 마", + "cxc-pabcd 쓰지 말고 그냥 고쳐줘", + "Don't use `cxc-loop` here", + ]) { + assert.equal(detectTrigger(prompt), null, prompt); + assert.equal(detectLoopArmRequest(prompt), false, prompt); + const cwd = freshCwd(); + try { + writeState(cwd, defaultState("negated")); + assert.equal(handleUserPromptSubmit(ups(prompt, cwd, "negated", "t1")), "", prompt); + assert.equal(readState(cwd, "negated").loopArmSeen, false, prompt); + } finally { rmSync(cwd, { recursive: true, force: true }); } + } + // A later negated clause does not cancel an earlier explicit request. + assert.equal(detectLoopArmRequest("Run cxc-loop for this task. Do not push."), true); + assert.equal(detectTrigger("Use cxc-pabcd to plan, not build"), "P"); + assert.equal(detectTrigger("PABCD로, 단계별로 진행해"), "P"); +}); + + test("detectTrigger: phase priority applies only within an explicit request line", () => { assert.equal(detectTrigger("Use cxc-pabcd to start Interview then Plan phase"), "I"); assert.equal(detectTrigger("Summarize interview notes\nUse cxc-pabcd to start Plan phase"), "P"); From 1b3fed38e7b262854915ca5c61cd6df2f6997744 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 01:14:00 +0900 Subject: [PATCH 29/90] fix(pabcd-state): scope request negation to the codexclaw action (#250) --- .../components/pabcd-state/dist/hook.js | 16 +++++++++++----- .../codexclaw/components/pabcd-state/src/hook.ts | 16 +++++++++++----- .../components/pabcd-state/test/hook.test.ts | 9 +++++++++ 3 files changed, 31 insertions(+), 10 deletions(-) diff --git a/plugins/codexclaw/components/pabcd-state/dist/hook.js b/plugins/codexclaw/components/pabcd-state/dist/hook.js index 3aef912c..a3c76916 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/hook.js +++ b/plugins/codexclaw/components/pabcd-state/dist/hook.js @@ -253,18 +253,24 @@ function requestLines(prompt ) { .replace(/"(?:\\.|[^"\\])*"|“[^”]*”|(? { "cxc-loop 돌리지 마", "cxc-pabcd 쓰지 말고 그냥 고쳐줘", "Don't use `cxc-loop` here", + "Please never run cxc-loop", + "cxc-loop 쓰지마", + "cxc-loop 말고 그냥 고쳐줘", ]) { assert.equal(detectTrigger(prompt), null, prompt); assert.equal(detectLoopArmRequest(prompt), false, prompt); @@ -102,6 +105,12 @@ test("issue 250: negated requests never arm or inject", () => { assert.equal(detectLoopArmRequest("Run cxc-loop for this task. Do not push."), true); assert.equal(detectTrigger("Use cxc-pabcd to plan, not build"), "P"); assert.equal(detectTrigger("PABCD로, 단계별로 진행해"), "P"); + // A constraint on another action keeps the explicit request. + assert.equal(detectLoopArmRequest("Run cxc-loop without asking me"), true); + assert.equal(detectTrigger("Use cxc-pabcd to plan without interviewing"), "P"); + assert.equal(detectLoopArmRequest("cxc-loop으로 끝까지 마무리해"), true); + assert.equal(detectLoopArmRequest("cxc-loop으로 끝까지 해줘, 푸시는 하지 마"), true); + assert.equal(detectLoopArmRequest("cxc-loop 돌려줘. 커밋은 하지 말고"), true); }); From 011efa3e5fb8f1163bc89d19d1b63213b2c31fe1 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 01:14:23 +0900 Subject: [PATCH 30/90] docs: publish measured test count (3650) --- README.ko.md | 2 +- README.md | 2 +- README.zh.md | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/README.ko.md b/README.ko.md index 7bd91054..2e32567a 100644 --- a/README.ko.md +++ b/README.ko.md @@ -13,7 +13,7 @@

CI - 3,649 tests + 3,650 tests 29 skills 29 hooks Documentation diff --git a/README.md b/README.md index 428fa714..bcf9d54e 100644 --- a/README.md +++ b/README.md @@ -13,7 +13,7 @@

CI - 3,649 tests + 3,650 tests 29 skills 29 hooks Documentation diff --git a/README.zh.md b/README.zh.md index 9975d663..d78a7178 100644 --- a/README.zh.md +++ b/README.zh.md @@ -13,7 +13,7 @@

CI - 3,649 tests + 3,650 tests 29 skills 29 hooks Documentation From 19ceb0569c51d6ce172150949165b071abbb3328 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 01:16:08 +0900 Subject: [PATCH 31/90] fix(pabcd-state): recognize indirect refusals and split a refused mode from a requested one (#250) --- plugins/codexclaw/components/pabcd-state/dist/hook.js | 9 ++++++--- plugins/codexclaw/components/pabcd-state/src/hook.ts | 9 ++++++--- .../codexclaw/components/pabcd-state/test/hook.test.ts | 7 +++++++ 3 files changed, 19 insertions(+), 6 deletions(-) diff --git a/plugins/codexclaw/components/pabcd-state/dist/hook.js b/plugins/codexclaw/components/pabcd-state/dist/hook.js index a3c76916..5e114021 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/hook.js +++ b/plugins/codexclaw/components/pabcd-state/dist/hook.js @@ -257,7 +257,7 @@ function requestLines(prompt ) { // "cxc-loop 돌리지 마") is not a request; a constraint on something else // ("run cxc-loop without asking me") is. Clauses split on sentence ends and // contrast words, never on commas. - for (const clause of unquoted.split(/[.;!?]\s*|\s+but\s+|\s*(?:하지만|그런데)\s*/i)) { + for (const clause of unquoted.split(/[.;!?]\s*|,\s*(?=(?:please\s+)?(?:use|run|start|invoke)\b)|\s+but\s+|\s*(?:하지만|그런데)\s*/i)) { const text = clause.trim(); if (text && !NEGATED_LEAD.test(text) && !NEGATED_TAIL.test(text)) result.push(text); } @@ -265,9 +265,12 @@ function requestLines(prompt ) { return result; } -/** English negation that governs the clause's own verb: "do not run ...", "never use ...". */ +/** + * English negation that governs the clause's own verb: "do not run ...", "never use ...", + * and indirect refusals such as "I don't want you to run ..." or "please do not ...". + */ const NEGATED_LEAD = - /^(?:(?:please|좀)\s+)?(?:do\s+not|don't|dont|never|no\s+need\s+to|avoid)\b/i; + /^(?:(?:please|좀)\s+)?(?:(?:i|we)\s+(?:do\s+not|don't|dont)\s+(?:want|need)\b|(?:i'd|i\s+would|we'd)\s+rather\s+not\b|do\s+not|don't|dont|never|no\s+need\s+to|avoid|stop)\b/i; /** Korean negation attached to the mode verb right after the marker: "cxc-loop 돌리지 마", "쓰지 말고". */ const NEGATED_TAIL = /(?:cxc-?(?:loop|pabcd)|pabcd)\S*\s*(?:을|를|은|는)?\s*(?:(?:돌리|쓰|사용하|실행하|켜|하)지\s*(?:마|말)|말고|금지)/i; diff --git a/plugins/codexclaw/components/pabcd-state/src/hook.ts b/plugins/codexclaw/components/pabcd-state/src/hook.ts index bcfd861b..fad1dfc5 100644 --- a/plugins/codexclaw/components/pabcd-state/src/hook.ts +++ b/plugins/codexclaw/components/pabcd-state/src/hook.ts @@ -257,7 +257,7 @@ function requestLines(prompt: string): string[] { // "cxc-loop 돌리지 마") is not a request; a constraint on something else // ("run cxc-loop without asking me") is. Clauses split on sentence ends and // contrast words, never on commas. - for (const clause of unquoted.split(/[.;!?]\s*|\s+but\s+|\s*(?:하지만|그런데)\s*/i)) { + for (const clause of unquoted.split(/[.;!?]\s*|,\s*(?=(?:please\s+)?(?:use|run|start|invoke)\b)|\s+but\s+|\s*(?:하지만|그런데)\s*/i)) { const text = clause.trim(); if (text && !NEGATED_LEAD.test(text) && !NEGATED_TAIL.test(text)) result.push(text); } @@ -265,9 +265,12 @@ function requestLines(prompt: string): string[] { return result; } -/** English negation that governs the clause's own verb: "do not run ...", "never use ...". */ +/** + * English negation that governs the clause's own verb: "do not run ...", "never use ...", + * and indirect refusals such as "I don't want you to run ..." or "please do not ...". + */ const NEGATED_LEAD = - /^(?:(?:please|좀)\s+)?(?:do\s+not|don't|dont|never|no\s+need\s+to|avoid)\b/i; + /^(?:(?:please|좀)\s+)?(?:(?:i|we)\s+(?:do\s+not|don't|dont)\s+(?:want|need)\b|(?:i'd|i\s+would|we'd)\s+rather\s+not\b|do\s+not|don't|dont|never|no\s+need\s+to|avoid|stop)\b/i; /** Korean negation attached to the mode verb right after the marker: "cxc-loop 돌리지 마", "쓰지 말고". */ const NEGATED_TAIL = /(?:cxc-?(?:loop|pabcd)|pabcd)\S*\s*(?:을|를|은|는)?\s*(?:(?:돌리|쓰|사용하|실행하|켜|하)지\s*(?:마|말)|말고|금지)/i; diff --git a/plugins/codexclaw/components/pabcd-state/test/hook.test.ts b/plugins/codexclaw/components/pabcd-state/test/hook.test.ts index b4fb4f61..9f772480 100644 --- a/plugins/codexclaw/components/pabcd-state/test/hook.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/hook.test.ts @@ -91,6 +91,10 @@ test("issue 250: negated requests never arm or inject", () => { "Please never run cxc-loop", "cxc-loop 쓰지마", "cxc-loop 말고 그냥 고쳐줘", + "I don't want you to run cxc-loop", + "I don't want you to use cxc-pabcd to plan", + "I'd rather not use cxc-loop", + "Stop using cxc-loop", ]) { assert.equal(detectTrigger(prompt), null, prompt); assert.equal(detectLoopArmRequest(prompt), false, prompt); @@ -111,6 +115,9 @@ test("issue 250: negated requests never arm or inject", () => { assert.equal(detectLoopArmRequest("cxc-loop으로 끝까지 마무리해"), true); assert.equal(detectLoopArmRequest("cxc-loop으로 끝까지 해줘, 푸시는 하지 마"), true); assert.equal(detectLoopArmRequest("cxc-loop 돌려줘. 커밋은 하지 말고"), true); + // A refused mode and a requested mode in one sentence: only the requested one counts. + assert.equal(detectLoopArmRequest("Don't run cxc-loop, use cxc-pabcd to plan instead"), false); + assert.equal(detectTrigger("Don't run cxc-loop, use cxc-pabcd to plan instead"), "P"); }); From c12343a7cb6499b35cdc0bcaa0c694df65a7c08a Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 01:17:21 +0900 Subject: [PATCH 32/90] fix(pabcd-state): treat "can you not" and "prefer not to" as refusals (#250) --- plugins/codexclaw/components/pabcd-state/dist/hook.js | 2 +- plugins/codexclaw/components/pabcd-state/src/hook.ts | 2 +- plugins/codexclaw/components/pabcd-state/test/hook.test.ts | 2 ++ 3 files changed, 4 insertions(+), 2 deletions(-) diff --git a/plugins/codexclaw/components/pabcd-state/dist/hook.js b/plugins/codexclaw/components/pabcd-state/dist/hook.js index 5e114021..e95713a6 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/hook.js +++ b/plugins/codexclaw/components/pabcd-state/dist/hook.js @@ -270,7 +270,7 @@ function requestLines(prompt ) { * and indirect refusals such as "I don't want you to run ..." or "please do not ...". */ const NEGATED_LEAD = - /^(?:(?:please|좀)\s+)?(?:(?:i|we)\s+(?:do\s+not|don't|dont)\s+(?:want|need)\b|(?:i'd|i\s+would|we'd)\s+rather\s+not\b|do\s+not|don't|dont|never|no\s+need\s+to|avoid|stop)\b/i; + /^(?:(?:please|좀)\s+)?(?:(?:i|we)\s+(?:do\s+not|don't|dont)\s+(?:want|need)\b|(?:i'd|i\s+would|we'd|we\s+would)\s+(?:rather|prefer)\s+(?:not|you\s+not|you\s+didn't)\b|(?:can|could|would|will)\s+you\s+(?:not|please\s+not)\b|do\s+not|don't|dont|never|no\s+need\s+to|avoid|stop)\b/i; /** Korean negation attached to the mode verb right after the marker: "cxc-loop 돌리지 마", "쓰지 말고". */ const NEGATED_TAIL = /(?:cxc-?(?:loop|pabcd)|pabcd)\S*\s*(?:을|를|은|는)?\s*(?:(?:돌리|쓰|사용하|실행하|켜|하)지\s*(?:마|말)|말고|금지)/i; diff --git a/plugins/codexclaw/components/pabcd-state/src/hook.ts b/plugins/codexclaw/components/pabcd-state/src/hook.ts index fad1dfc5..a63eb895 100644 --- a/plugins/codexclaw/components/pabcd-state/src/hook.ts +++ b/plugins/codexclaw/components/pabcd-state/src/hook.ts @@ -270,7 +270,7 @@ function requestLines(prompt: string): string[] { * and indirect refusals such as "I don't want you to run ..." or "please do not ...". */ const NEGATED_LEAD = - /^(?:(?:please|좀)\s+)?(?:(?:i|we)\s+(?:do\s+not|don't|dont)\s+(?:want|need)\b|(?:i'd|i\s+would|we'd)\s+rather\s+not\b|do\s+not|don't|dont|never|no\s+need\s+to|avoid|stop)\b/i; + /^(?:(?:please|좀)\s+)?(?:(?:i|we)\s+(?:do\s+not|don't|dont)\s+(?:want|need)\b|(?:i'd|i\s+would|we'd|we\s+would)\s+(?:rather|prefer)\s+(?:not|you\s+not|you\s+didn't)\b|(?:can|could|would|will)\s+you\s+(?:not|please\s+not)\b|do\s+not|don't|dont|never|no\s+need\s+to|avoid|stop)\b/i; /** Korean negation attached to the mode verb right after the marker: "cxc-loop 돌리지 마", "쓰지 말고". */ const NEGATED_TAIL = /(?:cxc-?(?:loop|pabcd)|pabcd)\S*\s*(?:을|를|은|는)?\s*(?:(?:돌리|쓰|사용하|실행하|켜|하)지\s*(?:마|말)|말고|금지)/i; diff --git a/plugins/codexclaw/components/pabcd-state/test/hook.test.ts b/plugins/codexclaw/components/pabcd-state/test/hook.test.ts index 9f772480..93582031 100644 --- a/plugins/codexclaw/components/pabcd-state/test/hook.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/hook.test.ts @@ -95,6 +95,8 @@ test("issue 250: negated requests never arm or inject", () => { "I don't want you to use cxc-pabcd to plan", "I'd rather not use cxc-loop", "Stop using cxc-loop", + "Can you not run cxc-loop?", + "I would prefer not to use cxc-pabcd for planning", ]) { assert.equal(detectTrigger(prompt), null, prompt); assert.equal(detectLoopArmRequest(prompt), false, prompt); From 4ee11120a356f9aaef97945ddc0a57d874c3c2cb Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 01:21:56 +0900 Subject: [PATCH 33/90] fix(pabcd-state): cxc-loop resume and continue requests arm after clause split (#250) --- plugins/codexclaw/components/pabcd-state/dist/hook.js | 2 +- plugins/codexclaw/components/pabcd-state/src/hook.ts | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/plugins/codexclaw/components/pabcd-state/dist/hook.js b/plugins/codexclaw/components/pabcd-state/dist/hook.js index e95713a6..d47d7258 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/hook.js +++ b/plugins/codexclaw/components/pabcd-state/dist/hook.js @@ -314,7 +314,7 @@ export function detectAgbrowseSearchRequest(prompt ) { export function detectLoopArmRequest(prompt ) { for (const line of requestLines(prompt)) { const marker = /\bcxc-?loop\b|\bcodexclaw:cxc-loop\b|\[\$?cxc-loop\]\(skill:\/\/[^)]+\)|\bgoal\s*plan\b|\bgoalplan\b|골플랜|\bhotl\b|(? Date: Mon, 28 Sep 2026 01:30:12 +0900 Subject: [PATCH 34/90] docs(plan): treat queued hosted CI as a wp5 external wait --- devlog/_plan/260927_issue_train/040_wp5_delivery.md | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/devlog/_plan/260927_issue_train/040_wp5_delivery.md b/devlog/_plan/260927_issue_train/040_wp5_delivery.md index 73d0d0a4..e3766299 100644 --- a/devlog/_plan/260927_issue_train/040_wp5_delivery.md +++ b/devlog/_plan/260927_issue_train/040_wp5_delivery.md @@ -1,5 +1,10 @@ # wp5 — Delivery and issue disposition + + +## Amendment 2026-09-28: CI queue as an external wait + +At wp2's C, PR #269's hosted CI sat queued because the account's Actions concurrency was held by another repository (lidge-jun/opencodex: 81 queued, 5 in progress at 16:30 UTC). Cancelling another repository's runs is outside this train's authority. The per-phase "CI, merge" step therefore moves to wp5 as an external wait: each implementation phase closes at D once its PR is open with local gates green, the next phase branches from the previous phase branch, and wp5 verifies hosted CI on every PR head, merges them into dev in order (retargeting each later PR to dev after its predecessor merges), and only then meets criterion c-7. This phase lands nothing new; it proves the merged state and records the issue decisions. ## Per implementation phase (wp2, wp3, wp4) @@ -21,4 +26,3 @@ This phase lands nothing new; it proves the merged state and records the issue d ## Acceptance All goalplan criteria met with captured evidence; `cxc loop validate` passes; `origin/dev` contains the three merge commits; no PR into `main` was opened by this unit. - From 29e34de84538adbaff04a8e7954323f4e6bc2eef Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 01:37:44 +0900 Subject: [PATCH 35/90] test(pabcd-state): resolve the CLI entry with fileURLToPath for Windows (#252) --- .../components/pabcd-state/test/subagent-evidence.test.ts | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/plugins/codexclaw/components/pabcd-state/test/subagent-evidence.test.ts b/plugins/codexclaw/components/pabcd-state/test/subagent-evidence.test.ts index 0fc2991d..0091543f 100644 --- a/plugins/codexclaw/components/pabcd-state/test/subagent-evidence.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/subagent-evidence.test.ts @@ -934,9 +934,9 @@ test("#252: policy-off dispatch leaves armed executor and worker evidence untouc try { writeFileSync(join(cwd, "codexclaw.json"), '{"pabcd":{"enabled":false}}'); writeState(cwd, { ...defaultState("s1"), phase: "B", orchestrationActive: true }); - const entry = new URL("../src/cli.ts", import.meta.url); + const entry = fileURLToPath(new URL("../src/cli.ts", import.meta.url)); for (const agent_type of ["executor", "worker"]) { - const result = spawnSync(process.execPath, [entry.pathname, "hook", "subagent-stop"], { + const result = spawnSync(process.execPath, [entry, "hook", "subagent-stop"], { input: JSON.stringify(payload(cwd, { agent_type, agent_id: agent_type })), encoding: "utf8", env: { ...process.env, CODEXCLAW_PABCD: "off" }, }); From 9c08e3140c6cc3fa7809832e856f0dc72659c8fd Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 01:38:03 +0900 Subject: [PATCH 36/90] docs(plan): re-anchor wp3 after wp2 and narrow MCP names --- .../020_wp3_agent_thread_permissions.md | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md b/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md index 634acb0e..71e5db2c 100644 --- a/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md +++ b/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md @@ -285,3 +285,14 @@ Cross-phase integration test (required, in `agent-thread-permissions.test.ts`): ## Out of scope No upstream Codex patch, runtime permission-profile change, automatic global opt-in, repo-local permission setting, forked-thread allowance, trust-state forging, or network/sandbox widening. `021_wp3_dispatch_guidance.md` owns the separate dispatch and #265 checkpoint text. + + +## wp3 re-verification against codex/issue-train-wp2 (supersedes stale anchors above) + +Architect handle `01a0e3b8-436b-7203-a4f6-97d24b865814` re-checked this plan after wp2 landed (HEAD 5178e261). Binding corrections: + +- **CLI placement (W3-2):** dispatch both verbs in `pabcd-state/src/cli.ts` right after stdin overflow handling and `const raw = stdin.raw` (`cli.ts:341-347`), before the hook recorder (`:348`), the subagent early exit (`:399-401`) and the PABCD switch (`:403-412`). On stdin overflow both verbs exit 0 with empty stdout. The generic SessionStart handler is now at `cli.ts:426-428`; `config interview` is at `cli.ts:217-252`. +- **Switch (W3-3):** neither `permission-request` nor `session-start-permission-advisory` is added to `PABCD_DISABLED_EVENTS` (`cli.ts:56-61`). +- **MCP names (W3-4, AD-4):** `coveredTool` accepts an MCP name only when it matches `/^mcp__[^_](?:[^]*?[^_])?__[^_].*$/` style two nonempty segments, implemented as `/^mcp__(.+?)__(.+)$/` with both captures nonempty after trimming underscores. Tests: `mcp__codex_app__create_thread` allowed; `mcp__`, `mcp__server`, `mcp____tool`, `mcp__server__` get no decision. +- **Switch tests (W3-5):** the built-CLI integration test runs both verbs twice, once with `CODEXCLAW_PABCD=off` and once with project `codexclaw.json` `{"pabcd":{"enabled":false}}`, each in a fresh cwd, asserting the allow bytes, the advisory JSON and no `/.codexclaw`. +- **Counts:** manifest hooks 29 -> 31 (`plugin.json:22-52`); test baseline 3650 before this phase; README hook badges at line 18, tests badges at line 16. From 25608a3c64d004d7297f3a918ba08e1d853d0131 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 01:38:50 +0900 Subject: [PATCH 37/90] docs(plan): make the wp3 module code carry the narrowed MCP rule --- .../020_wp3_agent_thread_permissions.md | 16 +++++++++++----- 1 file changed, 11 insertions(+), 5 deletions(-) diff --git a/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md b/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md index 71e5db2c..d10556cf 100644 --- a/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md +++ b/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md @@ -121,7 +121,13 @@ function codexConfigFullAccess(env: NodeJS.ProcessEnv): boolean { function coveredTool(name: unknown): boolean { return typeof name === "string" && (["Bash", "write_stdin", "apply_patch", "request_permissions"].includes(name) || - /^mcp__/.test(name)); + isMcpToolName(name)); +} + +/** `mcp____` with both segments present. */ +function isMcpToolName(name: string): boolean { + const match = /^mcp__(.+?)__(.+)$/.exec(name); + return !!match && match[1].replace(/_/g, "") !== "" && match[2].replace(/_/g, "") !== ""; } function parseHook(raw: string, event: string): JsonObject | null { @@ -256,7 +262,7 @@ Use `node:test` and `node:assert/strict`, as in adjacent component tests. Fixtur | `rejects non-default permission and subagent payloads` | `permission_mode` missing/`full-access` and present `agent_id` or `agent_type` (including empty string or null) each produce empty output. An inherited child must never use this exception. | | `rejects forked and user thread sources` | First-line `thread_source:"user"`, `"subagent"`, and `thread_source:"agent_created_thread"` with non-null `forked_from_id` all produce empty output, even with opt-in. Fork provenance remains outside this exception until measured. | | `rejects unknown or conflicting Codex config` | Missing TOML, malformed/torn required assignment, malformed top-level line, duplicate required key, missing either key, `on-request`, `workspace-write`, top-level `profile = "team"`, and quoted top-level `"profile" = "team"` each produce empty output. The global opt-in cannot override unknown effective policy. | -| `ignores project-local opt-in and noncovered tool` | Write a project `codexclaw.json`/`.codexclaw/config.json` true but omit global opt-in: empty; set relative `CODEXCLAW_HOME` or `CODEX_HOME`: empty; with absolute global paths and opt-in true, `Edit`, `mcp_tool`, `functions.exec`, and empty tool name: empty. The hook matcher is broad but the handler is not. | +| `ignores project-local opt-in and noncovered tool` | Write a project `codexclaw.json`/`.codexclaw/config.json` true but omit global opt-in: empty; set relative `CODEXCLAW_HOME` or `CODEX_HOME`: empty; with absolute global paths and opt-in true, `Edit`, `mcp_tool`, `functions.exec`, empty tool name, `mcp__`, `mcp__server`, `mcp____tool` and `mcp__server__`: empty; `mcp__codex_app__create_thread`: allow. The hook matcher is broad but the handler is not. | | `emits exact allow JSON and otherwise no stdout` | Parse the positive output and assert only `hookSpecificOutput.hookEventName`/`decision.behavior` keys; assert the output bytes equal the literal JSON above with no trailing LF. Test malformed hook JSON, missing/wrong `hook_event_name`, and thrown read errors yield `""`; no `deny`, `continue:false`, or stderr path exists. | | `advises agent-created default thread without opt-in` | Remove global opt-in, call SessionStart handler, parse output, assert nonempty `systemMessage` mentioning composer Full Access and opt-in, and `hookSpecificOutput = {hookEventName:"SessionStart",additionalContext:}` with `network` and `git` in the context. This proves AD-5 is independent of AD-2. | | `advisory is silent outside degraded agent-created context` | `approval_policy:on-request`, `sandbox_mode:workspace-write`, profile present, `permission_mode` other than default, `thread_source:user`, bad first line, and present agent field each yield `""`. Do not tell ordinary threads they degraded. | @@ -270,11 +276,11 @@ The existing real-hook golden tests at lines 53-76 and `EVENT_LABELS.PermissionR ## Runtime decision and bypass record -**Tier:** the synchronous PermissionRequest allow is a host permission decision **outside** structure/40's E1-E8 ladder (E1 means PreToolUse deny, not PermissionRequest allow); the SessionStart advisory is E4 context, and source tests/trust checks are E8. An allow does not override a separate denial. **Executing surface:** `handleAgentThreadPermissionRequest` and `handleAgentThreadSessionStartAdvisory` through their trusted hook manifests and `pabcd-state` CLI dispatch. **Known bypass:** an untrusted or absent hook has no effect; other hooks can deny; a direct caller can bypass the module predicates. **Residual risk:** runtime permission mode and sandbox/network constraints may differ from the top-level `config.toml` heuristic, and the `^mcp__` name filter may misclassify tools. **Wording downgrade:** “may suppress this approval prompt when all predicates hold,” not “agent threads have full access.” **Final enforcement layer:** Codex's PermissionRequest decision aggregator (`/tmp/cxc-perm/codex-src/codex-rs/hooks/src/events/permission_request.rs:149-169,301-327`); the advisory has no enforcement layer beyond context delivery. E8 tests check hook trust, allow bytes, and fail-open cases. +**Tier:** the synchronous PermissionRequest allow is a host permission decision **outside** structure/40's E1-E8 ladder (E1 means PreToolUse deny, not PermissionRequest allow); the SessionStart advisory is E4 context, and source tests/trust checks are E8. An allow does not override a separate denial. **Executing surface:** `handleAgentThreadPermissionRequest` and `handleAgentThreadSessionStartAdvisory` through their trusted hook manifests and `pabcd-state` CLI dispatch. **Known bypass:** an untrusted or absent hook has no effect; other hooks can deny; a direct caller can bypass the module predicates. **Residual risk:** runtime permission mode and sandbox/network constraints may differ from the top-level `config.toml` heuristic, and the `mcp____` name convention may misclassify tools. **Wording downgrade:** “may suppress this approval prompt when all predicates hold,” not “agent threads have full access.” **Final enforcement layer:** Codex's PermissionRequest decision aggregator (`/tmp/cxc-perm/codex-src/codex-rs/hooks/src/events/permission_request.rs:149-169,301-327`); the advisory has no enforcement layer beyond context delivery. E8 tests check hook trust, allow bytes, and fail-open cases. The PermissionRequest surface can suppress a user approval prompt only after all predicates pass. An empty stdout and exit 0 is no decision, so Codex keeps its normal prompt path; this is supported by `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/events/permission_request.rs:205-211`. The exact allow is parsed at `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/engine/output_parser.rs:184-204`. Never emit a denial, exit 2, `updatedInput`, `updatedPermissions`, or `interrupt`: the latter fields are unsupported at `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/schema.rs:199-217`, and exit 2 can deny at `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/events/permission_request.rs:249-263`. Another PermissionRequest hook's deny wins over this allow at `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/events/permission_request.rs:149-169,301-327`. -Residual risks to record in the implementation PR: top-level `config.toml` can differ from runtime overrides; `permission_mode:"default"` does not prove the active sandbox; an allow result does not widen filesystem or network permissions; the `^mcp__` name check is a naming heuristic; another hook may deny; a new/modified hook declaration changes its trust hash and needs re-approval. The advisory therefore says only that approval mode may have degraded. This workaround does not assert upstream #33282, #40793, or #41167 is fixed. +Residual risks to record in the implementation PR: top-level `config.toml` can differ from runtime overrides; `permission_mode:"default"` does not prove the active sandbox; an allow result does not widen filesystem or network permissions; the `mcp____` name check is a naming convention, not proof that the approval is an MCP call; another hook may deny; a new/modified hook declaration changes its trust hash and needs re-approval. The advisory therefore says only that approval mode may have degraded. This workaround does not assert upstream #33282, #40793, or #41167 is fixed. ## Verification and activation @@ -293,6 +299,6 @@ Architect handle `01a0e3b8-436b-7203-a4f6-97d24b865814` re-checked this plan aft - **CLI placement (W3-2):** dispatch both verbs in `pabcd-state/src/cli.ts` right after stdin overflow handling and `const raw = stdin.raw` (`cli.ts:341-347`), before the hook recorder (`:348`), the subagent early exit (`:399-401`) and the PABCD switch (`:403-412`). On stdin overflow both verbs exit 0 with empty stdout. The generic SessionStart handler is now at `cli.ts:426-428`; `config interview` is at `cli.ts:217-252`. - **Switch (W3-3):** neither `permission-request` nor `session-start-permission-advisory` is added to `PABCD_DISABLED_EVENTS` (`cli.ts:56-61`). -- **MCP names (W3-4, AD-4):** `coveredTool` accepts an MCP name only when it matches `/^mcp__[^_](?:[^]*?[^_])?__[^_].*$/` style two nonempty segments, implemented as `/^mcp__(.+?)__(.+)$/` with both captures nonempty after trimming underscores. Tests: `mcp__codex_app__create_thread` allowed; `mcp__`, `mcp__server`, `mcp____tool`, `mcp__server__` get no decision. +- **MCP names (W3-4, AD-4):** `coveredTool` accepts an MCP name only through `isMcpToolName` (the module code above now contains it): `/^mcp__(.+?)__(.+)$/` with both captures nonempty after removing underscores. Tests: `mcp__codex_app__create_thread` allowed; `mcp__`, `mcp__server`, `mcp____tool`, `mcp__server__` get no decision. - **Switch tests (W3-5):** the built-CLI integration test runs both verbs twice, once with `CODEXCLAW_PABCD=off` and once with project `codexclaw.json` `{"pabcd":{"enabled":false}}`, each in a fresh cwd, asserting the allow bytes, the advisory JSON and no `/.codexclaw`. - **Counts:** manifest hooks 29 -> 31 (`plugin.json:22-52`); test baseline 3650 before this phase; README hook badges at line 18, tests badges at line 16. From f15f0cc8d539f7fcdd097155a9bb4431a6429fe3 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 01:46:03 +0900 Subject: [PATCH 38/90] docs(plan): fold wp3 audit round 1 (network scope, TOML headers, request_permissions, project opt-in) --- .../020_wp3_agent_thread_permissions.md | 28 +++++++++++++------ 1 file changed, 20 insertions(+), 8 deletions(-) diff --git a/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md b/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md index d10556cf..40bb4ffa 100644 --- a/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md +++ b/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md @@ -1,12 +1,12 @@ # wp3 — Agent-created thread permission hook and advisory -Codex Desktop can start a `create_thread` child with approval prompts even when the user's top-level Codex configuration says `never` and `danger-full-access`. This phase offers a user-global, default-off PermissionRequest auto-allow for that exact provenance and a separate SessionStart warning that works without the opt-in. It does not claim to change the child's sandbox or network access. The symptom also occurs for projectless children, so no worktree predicate belongs in the eligibility test. +Codex Desktop can start a `create_thread` child with approval prompts even when the user's top-level Codex configuration says `never` and `danger-full-access`. This phase offers a user-global, default-off PermissionRequest auto-allow for that exact provenance and a separate SessionStart warning that works without the opt-in. It does not change the child's sandbox. With the opt-in it answers pending approvals the way the user's full-access config would, and that includes one-time network-access approvals, which Codex sends to hooks as `Bash` with a `network-access ` description (`codex-rs/core/src/tools/approvals.rs:217-224`); an allow there lets that one request through (`network_approval.rs:893-905`). The symptom also occurs for projectless children, so no worktree predicate belongs in the eligibility test. ## Phase contract - Class: C4, permission boundary. Binding decisions: AD-1 through AD-5 and AD-7 in `devlog/_plan/260927_issue_train/002_architect_consultation.md:9-15`. - Dependency: wp2's PABCD switch. The advisory is read-only and must not create `/.codexclaw` itself; the separate existing SessionStart bootstrapping hook still creates state and 011's `.gitignore` when PABCD policy is enabled. The module itself only reads files and returns JSON and does not call `handleSessionStart`; the shared CLI path still records a hook observation under `CODEX_HOME` (plugins/codexclaw/scripts/hook-observation.mjs:17,70), never under the cwd. Both verbs stay active when `CODEXCLAW_PABCD=off` or `pabcd.enabled=false`, because they are not PABCD policy; 012's switch must not list them. -- Success: with the explicit global opt-in, an agent-created root thread whose hook says `permission_mode: "default"` and whose user Codex config shows explicit top-level evidence `approval_policy = "never"` and `sandbox_mode = "danger-full-access"` (no profile) emits the exact allow object for Bash, write_stdin, apply_patch, request_permissions, and tool names that follow the `mcp____` naming convention. This is config evidence of user intent, not proof of the thread's effective permission; an exact guarantee needs Codex to expose the resolved policy and sandbox in PermissionRequest input. Every missing, mismatched, corrupt or unknown input emits zero stdout bytes and exits 0. The advisory is independent of opt-in. +- Success: with the explicit global opt-in, an agent-created root thread whose hook says `permission_mode: "default"` and whose user Codex config shows explicit top-level evidence `approval_policy = "never"` and `sandbox_mode = "danger-full-access"` (no profile) emits the exact allow object for Bash (including one-time network-access approvals), write_stdin, apply_patch, and tool names that follow the `mcp____` naming convention. This is config evidence of user intent, not proof of the thread's effective permission; an exact guarantee needs Codex to expose the resolved policy and sandbox in PermissionRequest input. Every missing, mismatched, corrupt or unknown input emits zero stdout bytes and exits 0. The advisory is independent of opt-in. - Runtime proof boundary: hook input has `session_id`, `transcript_path`, `permission_mode`, `tool_name`, and optional `agent_id`/`agent_type` at `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/schema.rs:298-318`; SessionStart input has the first three at `/tmp/cxc-perm/codex-src/codex-rs/hooks/src/schema.rs:496-510`. `SessionMeta` stores `id` and `thread_source` at `/tmp/cxc-perm/codex-src/codex-rs/protocol/src/protocol.rs:3128-3154`, and `ThreadSource::Feature` serializes its feature string at `/tmp/cxc-perm/codex-src/codex-rs/protocol/src/protocol.rs:2841-2857`. The specific `agent_created_thread` value is the observed rollout fixture from this issue train, not a universal enum variant. ## File change map @@ -26,7 +26,7 @@ The value must be the JSON boolean `true`; missing file/key, malformed JSON, arr ```ts import { closeSync, openSync, readFileSync, readSync, statSync } from "node:fs"; import { homedir } from "node:os"; -import { isAbsolute, join } from "node:path"; +import { isAbsolute, join, relative, resolve } from "node:path"; const MAX_META_LINE_BYTES = 64 * 1024; const MAX_CONFIG_BYTES = 1024 * 1024; @@ -79,10 +79,13 @@ function boundedText(path: string): string | null { return readFileSync(path, "utf8"); } -function globalOptIn(env: NodeJS.ProcessEnv): boolean { +function globalOptIn(env: NodeJS.ProcessEnv, cwd: string): boolean { const override = env.CODEXCLAW_HOME?.trim(); if (override && !isAbsolute(override)) return false; const home = override || join(homedir(), ".codexclaw"); + // A project cannot grant this: an override that resolves inside the session cwd is ignored. + const rel = relative(resolve(cwd), resolve(home)); + if (rel === "" || (!rel.startsWith("..") && !isAbsolute(rel))) return false; const config = object(JSON.parse(boundedText(join(home, "config.json")) ?? "null")); const permissions = object(config?.permissions); return permissions?.agentCreatedThreadAutoAllow === true; @@ -100,7 +103,7 @@ function codexConfigFullAccess(env: NodeJS.ProcessEnv): boolean { if (line === "" || line.startsWith("#")) continue; if (line.startsWith("[")) { // An invalid table header leaves the ownership of following keys unknown. - if (!/^\[\[?[^\]\r\n]+\]\]?\s*(?:#.*)?$/.test(line)) return false; + if (!/^(?:\[\[[^[\]\r\n]+\]\]|\[[^[\]\r\n]+\])\s*(?:#.*)?$/.test(line)) return false; break; } if (/^(?:profile|"profile"|'profile')\s*=/.test(line)) return false; @@ -120,7 +123,7 @@ function codexConfigFullAccess(env: NodeJS.ProcessEnv): boolean { function coveredTool(name: unknown): boolean { return typeof name === "string" && - (["Bash", "write_stdin", "apply_patch", "request_permissions"].includes(name) || + (["Bash", "write_stdin", "apply_patch"].includes(name) || isMcpToolName(name)); } @@ -140,7 +143,8 @@ export function handleAgentThreadPermissionRequest( ): string { try { const input = parseHook(raw, "PermissionRequest"); - return input && coveredTool(input.tool_name) && globalOptIn(env) && + return input && typeof input.cwd === "string" && input.cwd !== "" && + coveredTool(input.tool_name) && globalOptIn(env, input.cwd) && agentCreatedRoot(input) && codexConfigFullAccess(env) ? ALLOW : ""; } catch { return ""; @@ -256,7 +260,7 @@ Use `node:test` and `node:assert/strict`, as in adjacent component tests. Fixtur | Named test | Exact assertion and reason it fails before this phase | |---|---| -| `allows opted-in agent-created root thread for covered tools` | For `Bash`, `write_stdin`, `apply_patch`, `request_permissions`, `mcp__codex_app__create_worktree`, assert `strictEqual(result, '{"hookSpecificOutput":{"hookEventName":"PermissionRequest","decision":{"behavior":"allow"}}}')`. No handler exists before this phase. | +| `allows opted-in agent-created root thread for covered tools` | For `Bash`, a `Bash` network approval (`tool_input.description` = `network-access example.com`), `write_stdin`, `apply_patch`, `mcp__codex_app__create_worktree`, assert `strictEqual(result, '{"hookSpecificOutput":{"hookEventName":"PermissionRequest","decision":{"behavior":"allow"}}}')`. No handler exists before this phase. | | `default-off global setting leaves approval to Codex` | For missing `config.json`, missing `permissions`, malformed JSON, `permissions` array, `false`, and string `"true"`, assert empty output. The new gate must not silently grant approval. | | `rejects missing corrupt and mismatched rollout identity` | Subcases: missing/empty `session_id`, null/missing transcript path, nonexistent path, empty file, malformed first JSONL line, first line not `session_meta`, missing payload, missing id, wrong id, missing/wrong `thread_source`, first line >64 KiB, and a valid second line after a wrong first line; assert empty output each. Without first-record verification an arbitrary thread could be allowed. | | `rejects non-default permission and subagent payloads` | `permission_mode` missing/`full-access` and present `agent_id` or `agent_type` (including empty string or null) each produce empty output. An inherited child must never use this exception. | @@ -302,3 +306,11 @@ Architect handle `01a0e3b8-436b-7203-a4f6-97d24b865814` re-checked this plan aft - **MCP names (W3-4, AD-4):** `coveredTool` accepts an MCP name only through `isMcpToolName` (the module code above now contains it): `/^mcp__(.+?)__(.+)$/` with both captures nonempty after removing underscores. Tests: `mcp__codex_app__create_thread` allowed; `mcp__`, `mcp__server`, `mcp____tool`, `mcp__server__` get no decision. - **Switch tests (W3-5):** the built-CLI integration test runs both verbs twice, once with `CODEXCLAW_PABCD=off` and once with project `codexclaw.json` `{"pabcd":{"enabled":false}}`, each in a fresh cwd, asserting the allow bytes, the advisory JSON and no `/.codexclaw`. - **Counts:** manifest hooks 29 -> 31 (`plugin.json:22-52`); test baseline 3650 before this phase; README hook badges at line 18, tests badges at line 16. + + +## wp3 audit folds (round 1) + +- **Network approvals are in scope.** A full-access user config grants network, so the opt-in also answers `Bash` network-access approvals once. Tests: `network-access approval is allowed with opt-in` and `network-access approval gets no decision without opt-in`. +- **Malformed TOML fails closed.** Table headers must have paired delimiters (`[name]` or `[[name]]`); `[[profiles]`, `[profiles]]` and any other line starting with `[` return no decision. Tests: both malformed headers plus a valid `[features]` header after the two keys. +- **`request_permissions` is out of scope.** The current host routes it straight to Guardian without PermissionRequest hooks (`codex-rs/core/src/session/mod.rs:2985-3008`, `approvals.rs:866-868`), so listing it would be a coverage claim no live path exercises. It is removed from `coveredTool` and the tests; revisit if the host starts routing it through hooks. +- **A project cannot grant the opt-in.** The handler requires a string `cwd` in the hook input and ignores a `CODEXCLAW_HOME` that resolves inside that cwd. Test: absolute `CODEXCLAW_HOME=/.codexclaw` with `agentCreatedThreadAutoAllow: true` gets no decision; a sibling temp dir outside cwd with the same file allows. From b2a4b6001480603668963f4bfe44a00cca112ed0 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 02:07:40 +0900 Subject: [PATCH 39/90] feat(pabcd-state): handle agent-created thread permissions --- plugins/codexclaw/.codex-plugin/plugin.json | 2 + .../cxc-ops/test/hook-trust.test.ts | 27 ++ .../src/agent-thread-permissions.ts | 228 ++++++++++++++++ .../components/pabcd-state/src/cli.ts | 17 ++ .../test/agent-thread-permissions.test.ts | 248 ++++++++++++++++++ ...mission-request-allowing-agent-thread.json | 13 + ...art-advising-agent-thread-permissions.json | 12 + 7 files changed, 547 insertions(+) create mode 100644 plugins/codexclaw/components/pabcd-state/src/agent-thread-permissions.ts create mode 100644 plugins/codexclaw/components/pabcd-state/test/agent-thread-permissions.test.ts create mode 100644 plugins/codexclaw/hooks/permission-request-allowing-agent-thread.json create mode 100644 plugins/codexclaw/hooks/session-start-advising-agent-thread-permissions.json diff --git a/plugins/codexclaw/.codex-plugin/plugin.json b/plugins/codexclaw/.codex-plugin/plugin.json index 9661c198..d48e0fbe 100644 --- a/plugins/codexclaw/.codex-plugin/plugin.json +++ b/plugins/codexclaw/.codex-plugin/plugin.json @@ -24,9 +24,11 @@ "./hooks/session-start-bootstrapping-pabcd-state.json", "./hooks/session-start-healing-declared-features.json", "./hooks/session-start-announcing-map-affordance.json", + "./hooks/session-start-advising-agent-thread-permissions.json", "./hooks/user-prompt-submit-checking-pabcd-trigger.json", "./hooks/stop-checking-pabcd-continuation.json", "./hooks/pre-tool-use-guarding-goal-budget.json", + "./hooks/permission-request-allowing-agent-thread.json", "./hooks/pre-tool-use-guarding-interview-in-goal.json", "./hooks/pre-tool-use-guarding-goal-complete.json", "./hooks/post-tool-use-capturing-interview-answers.json", diff --git a/plugins/codexclaw/components/cxc-ops/test/hook-trust.test.ts b/plugins/codexclaw/components/cxc-ops/test/hook-trust.test.ts index 6a98c86e..3998d28f 100644 --- a/plugins/codexclaw/components/cxc-ops/test/hook-trust.test.ts +++ b/plugins/codexclaw/components/cxc-ops/test/hook-trust.test.ts @@ -75,6 +75,33 @@ test("identityHash keeps the matcher in the live SubagentStop hook golden fixtur ); }); +test("new agent thread hooks have stable trust identities and both require trust", () => { + const specs = [ + ["permission-request-allowing-agent-thread.json", "PermissionRequest", "permission_request"], + ["session-start-advising-agent-thread-permissions.json", "SessionStart", "session_start"], + ] as const; + const entries = listHookEntries(PLUGIN_ROOT, "codexclaw@local"); + const selected = specs.map(([file, event, label]) => { + const doc = JSON.parse(readFileSync(join(PLUGIN_ROOT, "hooks", file), "utf8")) as { + hooks: Record>; + }; + const group = doc.hooks[event][0]; + const hash = identityHash(event, group.matcher, group.hooks[0]); + assert.match(hash, /^sha256:[a-f0-9]{64}$/); + if (event === "PermissionRequest") assert.equal(group.matcher, "*"); + const matches = entries.filter((entry) => entry.key === `codexclaw@local:hooks/${file}:${label}:0:0`); + assert.equal(matches.length, 1); + assert.equal(matches[0].hash, hash); + return matches[0]; + }); + const home = makeCodexHome(""); + let statuses = diagnoseHookTrust(home, PLUGIN_ROOT, "codexclaw@local"); + assert.deepEqual(selected.map((entry) => statuses.find((item) => item.key === entry.key)?.status), ["untrusted", "untrusted"]); + writeFileSync(join(home, "config.toml"), `${trustSection(selected[0])}\n${trustSection(selected[1], "sha256:stale")}`); + statuses = diagnoseHookTrust(home, PLUGIN_ROOT, "codexclaw@local"); + assert.deepEqual(selected.map((entry) => statuses.find((item) => item.key === entry.key)?.status), ["trusted", "drifted"]); +}); + test("identityHash filters matcher by event", () => { const handler = command("echo ok"); assert.equal(identityHash("Stop", "^ignored$", handler), identityHash("Stop", undefined, handler)); diff --git a/plugins/codexclaw/components/pabcd-state/src/agent-thread-permissions.ts b/plugins/codexclaw/components/pabcd-state/src/agent-thread-permissions.ts new file mode 100644 index 00000000..c7a7ebe2 --- /dev/null +++ b/plugins/codexclaw/components/pabcd-state/src/agent-thread-permissions.ts @@ -0,0 +1,228 @@ +import { closeSync, lstatSync, openSync, readFileSync, readSync, realpathSync, statSync } from "node:fs"; +import { homedir } from "node:os"; +import { isAbsolute, join, relative, resolve } from "node:path"; + +const MAX_META_LINE_BYTES = 64 * 1024; +const MAX_CONFIG_BYTES = 1024 * 1024; +const ALLOW = '{"hookSpecificOutput":{"hookEventName":"PermissionRequest","decision":{"behavior":"allow"}}}'; +const MODEL_ADVICE = "This Codex Desktop agent-created thread may show approval prompts even though the user Codex config requests full access. Request escalation explicitly for network or git operations when needed. If the user enabled permissions.agentCreatedThreadAutoAllow, codexclaw answers pending approvals, including one-time network requests, without prompting; it never changes this thread's sandbox."; +const USER_ADVICE = "This agent-created thread started in the default approval mode despite your full-access Codex config, so approval prompts may appear. You can switch this thread to Full Access in the composer, or set permissions.agentCreatedThreadAutoAllow to true in ~/.codexclaw/config.json so codexclaw answers these approvals for you, including one-time network requests."; + +type JsonObject = Record; + +function object(value: unknown): JsonObject | null { + return value !== null && typeof value === "object" && !Array.isArray(value) + ? value as JsonObject : null; +} + +function readFirstRecord(path: string): JsonObject | null { + if (!isAbsolute(path)) return null; + const fd = openSync(path, "r"); + try { + const buffer = Buffer.alloc(MAX_META_LINE_BYTES + 1); + let used = 0; + while (used < buffer.length) { + const count = readSync(fd, buffer, used, buffer.length - used, used); + if (count === 0) break; + used += count; + const end = buffer.subarray(0, used).indexOf(10); + if (end >= 0) return end > MAX_META_LINE_BYTES ? null : + object(JSON.parse(buffer.subarray(0, end).toString("utf8"))); + } + if (used === 0 || used > MAX_META_LINE_BYTES) return null; + return object(JSON.parse(buffer.subarray(0, used).toString("utf8"))); + } finally { + closeSync(fd); + } +} + +function agentCreatedRoot(input: JsonObject): boolean { + if (input.permission_mode !== "default" || + typeof input.session_id !== "string" || input.session_id.length === 0 || + typeof input.transcript_path !== "string" || + Object.hasOwn(input, "agent_id") || Object.hasOwn(input, "agent_type")) return false; + const record = readFirstRecord(input.transcript_path); + const payload = object(record?.payload); + return record?.type === "session_meta" && payload?.id === input.session_id && + payload.thread_source === "agent_created_thread" && payload.forked_from_id == null; +} + +function boundedText(path: string): string | null { + const stat = statSync(path); + if (!stat.isFile() || stat.size > MAX_CONFIG_BYTES) return null; + return readFileSync(path, "utf8"); +} + +function globalOptIn(env: NodeJS.ProcessEnv, cwd: string): boolean { + const override = env.CODEXCLAW_HOME?.trim(); + if (override && !isAbsolute(override)) return false; + const home = override || join(homedir(), ".codexclaw"); + const configPath = join(home, "config.json"); + const realCwd = realpathSync(cwd); + const realConfig = realpathSync(configPath); + const rel = relative(realCwd, realConfig); + const withinCwd = rel === "" || (!rel.startsWith("..") && !isAbsolute(rel)); + const defaultAtHome = !override && realCwd === realpathSync(homedir()) && + !lstatSync(configPath).isSymbolicLink() && lstatSync(configPath).isFile(); + if (withinCwd && !defaultAtHome) return false; + const config = object(JSON.parse(boundedText(realConfig) ?? "null")); + const permissions = object(config?.permissions); + return permissions?.agentCreatedThreadAutoAllow === true; +} + +/** A bounded structural scanner: unknown TOML is never permission evidence. */ +function validValue(source: string): boolean { + let index = 0; + const scalar = /^(?:true|false|[+-]?(?:0|[1-9](?:[0-9_]*[0-9])?)(?:\.[0-9_]+)?(?:[eE][+-]?[0-9_]+)?|[+-]?(?:inf|nan)|\d{4}-\d\d-\d\d(?:[Tt ]\d\d:\d\d:\d\d(?:\.\d+)?(?:[Zz]|[+-]\d\d:\d\d)?)?|\d\d:\d\d:\d\d(?:\.\d+)?|0[xX][0-9a-fA-F_]+|0[oO][0-7_]+|0[bB][01_]+)$/; + const skip = (): void => { + while (index < source.length) { + if (/\s/.test(source[index])) { index += 1; continue; } + if (source[index] === "#") { + const end = source.indexOf("\n", index); + index = end < 0 ? source.length : end; + continue; + } + break; + } + }; + const quoted = (): boolean => { + const quote = source[index]; + const triple = source.startsWith(quote.repeat(3), index); + const mark = triple ? quote.repeat(3) : quote; + index += mark.length; + while (index < source.length) { + if (source.startsWith(mark, index)) { index += mark.length; return true; } + if (!triple && /[\r\n]/.test(source[index])) return false; + if (quote === '"' && source[index] === "\\") index += 1; + index += 1; + } + return false; + }; + const key = (): boolean => { + if (source[index] === '"' || source[index] === "'") return quoted(); + const match = /^[A-Za-z0-9_-]+/.exec(source.slice(index)); + if (!match) return false; + index += match[0].length; + return true; + }; + const value = (depth: number): boolean => { + if (depth > 64) return false; + skip(); + const opener = source[index]; + if (opener === '"' || opener === "'") return quoted(); + if (opener === "[" || opener === "{") { + index += 1; + const closer = opener === "[" ? "]" : "}"; + skip(); + if (source[index] === closer) { index += 1; return true; } + while (index < source.length) { + if (opener === "{") { + if (!key()) return false; + skip(); + if (source[index++] !== "=") return false; + } + if (!value(depth + 1)) return false; + skip(); + if (source[index] === closer) { index += 1; return true; } + if (source[index++] !== ",") return false; + skip(); + if (opener === "[" && source[index] === closer) { index += 1; return true; } + } + return false; + } + const bare = /^[^\s,\]\}#]+/.exec(source.slice(index)); + if (!bare || !scalar.test(bare[0])) return false; + index += bare[0].length; + return true; + }; + if (!value(0)) return false; + skip(); + return index === source.length; +} + +function validTomlAndTopLevel(content: string): boolean { + const seen = new Map(); + let inTopLevel = true; + const lines = content.replace(/^\uFEFF/, "").split(/\r?\n/); + for (let index = 0; index < lines.length; index += 1) { + const line = lines[index].trim(); + if (line === "" || line.startsWith("#")) continue; + if (line.startsWith("[")) { + if (!/^(?:\[\[[^[\]\r\n]+\]\]|\[[^[\]\r\n]+\])\s*(?:#.*)?$/.test(line)) return false; + inTopLevel = false; + continue; + } + const assignment = /^([A-Za-z0-9_-]+|"[^"\r\n]+"|'[^'\r\n]+')\s*=\s*(.*)$/.exec(line); + if (!assignment) return false; + const key = assignment[1].replace(/^["']|["']$/g, ""); + if (inTopLevel && key === "profile") return false; + let value = assignment[2]; + while (!validValue(value)) { + // Only arrays, inline tables and triple strings may continue onto another line. + if (!/^(?:\[|\{|"""|''')/.test(value) || index + 1 >= lines.length) return false; + value += `\n${lines[++index]}`; + } + if (inTopLevel && (key === "approval_policy" || key === "sandbox_mode")) { + if (seen.has(key)) return false; + const exact = /^"([^"\r\n]*)"\s*(?:#.*)?$/.exec(value); + if (!exact) return false; + seen.set(key, exact[1]); + } + } + return seen.get("approval_policy") === "never" && + seen.get("sandbox_mode") === "danger-full-access"; +} + +function codexConfigFullAccess(env: NodeJS.ProcessEnv): boolean { + const override = env.CODEX_HOME?.trim(); + if (override && !isAbsolute(override)) return false; + const home = override || join(homedir(), ".codex"); + const content = boundedText(join(home, "config.toml")); + if (content === null) return false; + return validTomlAndTopLevel(content); +} + +function coveredTool(name: unknown): boolean { + return typeof name === "string" && + (["Bash", "write_stdin", "apply_patch"].includes(name) || + isMcpToolName(name)); +} + +/** `mcp____` with both segments present. */ +function isMcpToolName(name: string): boolean { + const match = /^mcp__(.+?)__(.+)$/.exec(name); + return !!match && match[1].replace(/_/g, "") !== "" && match[2].replace(/_/g, "") !== ""; +} + +function parseHook(raw: string, event: string): JsonObject | null { + const input = object(JSON.parse(raw)); + return input?.hook_event_name === event ? input : null; +} + +export function handleAgentThreadPermissionRequest( + raw: string, env: NodeJS.ProcessEnv = process.env, +): string { + try { + const input = parseHook(raw, "PermissionRequest"); + return input && typeof input.cwd === "string" && input.cwd !== "" && + coveredTool(input.tool_name) && globalOptIn(env, input.cwd) && + agentCreatedRoot(input) && codexConfigFullAccess(env) ? ALLOW : ""; + } catch { + return ""; + } +} + +export function handleAgentThreadSessionStartAdvisory( + raw: string, env: NodeJS.ProcessEnv = process.env, +): string { + try { + const input = parseHook(raw, "SessionStart"); + if (!input || !agentCreatedRoot(input) || !codexConfigFullAccess(env)) return ""; + return `${JSON.stringify({ + systemMessage: USER_ADVICE, + hookSpecificOutput: { hookEventName: "SessionStart", additionalContext: MODEL_ADVICE }, + })}\n`; + } catch { + return ""; + } +} diff --git a/plugins/codexclaw/components/pabcd-state/src/cli.ts b/plugins/codexclaw/components/pabcd-state/src/cli.ts index 90cbaa9c..d88102ff 100644 --- a/plugins/codexclaw/components/pabcd-state/src/cli.ts +++ b/plugins/codexclaw/components/pabcd-state/src/cli.ts @@ -340,11 +340,28 @@ async function main(): Promise { const stdin = readStdin(); if (stdin.overflow) { + if (event === "permission-request" || event === "session-start-permission-advisory") { + process.exit(0); + } const denied = oversizedHookOutput(event); if (denied) process.stdout.write(denied); process.exit(denied ? 0 : 1); } const raw = stdin.raw; + if (event === "permission-request" || event === "session-start-permission-advisory") { + try { + recordHookInvocation(raw, "pabcd-state", event, import.meta.url); + const { handleAgentThreadPermissionRequest, handleAgentThreadSessionStartAdvisory } = + await import("./agent-thread-permissions.ts"); + const result = event === "permission-request" + ? handleAgentThreadPermissionRequest(raw) + : handleAgentThreadSessionStartAdvisory(raw); + if (result) process.stdout.write(result); + } catch { + // Fail open: no decision/advisory, exit 0. + } + process.exit(0); + } recordHookInvocation(raw, "pabcd-state", event, import.meta.url); let output = ""; diff --git a/plugins/codexclaw/components/pabcd-state/test/agent-thread-permissions.test.ts b/plugins/codexclaw/components/pabcd-state/test/agent-thread-permissions.test.ts new file mode 100644 index 00000000..36d66660 --- /dev/null +++ b/plugins/codexclaw/components/pabcd-state/test/agent-thread-permissions.test.ts @@ -0,0 +1,248 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { spawnSync } from "node:child_process"; +import { existsSync, mkdtempSync, mkdirSync, rmSync, symlinkSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { dirname, join, resolve } from "node:path"; +import { fileURLToPath } from "node:url"; +import { handleAgentThreadPermissionRequest as permission, handleAgentThreadSessionStartAdvisory as advisory } from "../src/agent-thread-permissions.ts"; + +const ALLOW = '{"hookSpecificOutput":{"hookEventName":"PermissionRequest","decision":{"behavior":"allow"}}}'; +const CLI = resolve(dirname(fileURLToPath(import.meta.url)), "../dist/cli.js"); +const FULL = 'approval_policy = "never"\nsandbox_mode = "danger-full-access"\n'; + +function fixture(t: { after: (fn: () => void) => void }) { + const root = mkdtempSync(join(tmpdir(), "cxc-agent-permission-")); + t.after(() => rmSync(root, { recursive: true, force: true })); + const cwd = join(root, "project"); + const codexHome = join(root, "codex"); + const clawHome = join(root, "global-claw"); + for (const dir of [cwd, codexHome, clawHome]) mkdirSync(dir); + const transcript = join(root, "rollout.jsonl"); + const meta = { type: "session_meta", payload: { id: "fixture-id", thread_source: "agent_created_thread" } }; + writeFileSync(transcript, `${JSON.stringify(meta)}\n`); + writeFileSync(join(codexHome, "config.toml"), FULL); + writeFileSync(join(clawHome, "config.json"), '{"permissions":{"agentCreatedThreadAutoAllow":true}}'); + const env: NodeJS.ProcessEnv = { CODEX_HOME: codexHome, CODEXCLAW_HOME: clawHome }; + const input = { hook_event_name: "PermissionRequest", cwd, session_id: "fixture-id", transcript_path: transcript, permission_mode: "default", tool_name: "Bash" }; + const send = (change: Record = {}, e: NodeJS.ProcessEnv = env) => permission(JSON.stringify({ ...input, ...change }), e); + const advise = (change: Record = {}, e: NodeJS.ProcessEnv = env) => advisory(JSON.stringify({ ...input, hook_event_name: "SessionStart", ...change }), e); + return { root, cwd, codexHome, clawHome, transcript, meta, env, input, send, advise }; +} + +function cli(verb: string, input: unknown, cwd: string, env: NodeJS.ProcessEnv) { + return spawnSync(process.execPath, [CLI, "hook", verb], { + cwd, env: { ...process.env, ...env }, input: typeof input === "string" ? input : JSON.stringify(input), encoding: "utf8", + }); +} + +test("allows opted-in agent-created root thread for covered tools", (t) => { + const f = fixture(t); + for (const tool_name of ["Bash", "write_stdin", "apply_patch", "mcp__codex_app__create_worktree", "mcp__codex_app__create_thread"]) { + assert.equal(f.send({ tool_name }), ALLOW, tool_name); + } + assert.equal(f.send({ tool_name: "Bash", tool_input: { description: "network-access example.com" } }), ALLOW); +}); + +test("network-access approval is allowed with opt-in", (t) => { + const f = fixture(t); + assert.equal(f.send({ tool_input: { description: "network-access example.com" } }), ALLOW); +}); + +test("network-access approval gets no decision without opt-in", (t) => { + const f = fixture(t); + writeFileSync(join(f.clawHome, "config.json"), "{}"); + assert.equal(f.send({ tool_input: { description: "network-access example.com" } }), ""); +}); + +test("default-off global setting leaves approval to Codex", (t) => { + const f = fixture(t); + const path = join(f.clawHome, "config.json"); + for (const value of [null, "{}", "{", '{"permissions":[]}', '{"permissions":{"agentCreatedThreadAutoAllow":false}}', '{"permissions":{"agentCreatedThreadAutoAllow":"true"}}']) { + if (value === null) rmSync(path, { force: true }); else writeFileSync(path, value); + assert.equal(f.send(), "", String(value)); + } +}); + +test("rejects missing corrupt and mismatched rollout identity", (t) => { + const f = fixture(t); + for (const change of [{ session_id: "" }, { session_id: null }, { session_id: undefined }, { transcript_path: null }, { transcript_path: undefined }, { transcript_path: "" }, { transcript_path: join(f.root, "absent") }, { transcript_path: "relative.jsonl" }]) { + assert.equal(f.send(change), "", JSON.stringify(change)); + } + const records = ["", "{bad\n", JSON.stringify({ type: "other", payload: f.meta.payload }), JSON.stringify({ type: "session_meta" }), ...[ + {}, { id: "wrong", thread_source: "agent_created_thread" }, { id: "fixture-id" }, { id: "fixture-id", thread_source: "user" }, + ].map((payload) => JSON.stringify({ type: "session_meta", payload })), + `{"padding":"${"x".repeat(65536)}"}`, `${JSON.stringify({ type: "other" })}\n${JSON.stringify(f.meta)}`]; + for (const value of records) { + writeFileSync(f.transcript, `${value}\n`); + assert.equal(f.send(), "", value.slice(0, 70)); + } +}); + +test("rejects non-default permission and subagent payloads", (t) => { + const f = fixture(t); + for (const change of [{ permission_mode: null }, { permission_mode: "full-access" }, { agent_id: "" }, { agent_id: null }, { agent_type: "" }, { agent_type: null }]) { + assert.equal(f.send(change), "", JSON.stringify(change)); + } +}); + +test("rejects forked and user thread sources", (t) => { + const f = fixture(t); + for (const payload of [{ id: "fixture-id", thread_source: "user" }, { id: "fixture-id", thread_source: "subagent" }, { id: "fixture-id", thread_source: "agent_created_thread", forked_from_id: "parent" }]) { + writeFileSync(f.transcript, `${JSON.stringify({ type: "session_meta", payload })}\n`); + assert.equal(f.send(), ""); + } +}); + +test("rejects unknown or conflicting Codex config", (t) => { + const f = fixture(t); + const path = join(f.codexHome, "config.toml"); + rmSync(path); + assert.equal(f.send(), ""); + for (const value of [ + 'approval_policy = "never\nsandbox_mode = "danger-full-access"', + `${FULL}invalid top-level`, `${FULL}approval_policy = "never"`, + 'approval_policy = "never"', 'sandbox_mode = "danger-full-access"', + 'approval_policy = "on-request"\nsandbox_mode = "danger-full-access"', + 'approval_policy = "never"\nsandbox_mode = "workspace-write"', + `${FULL}profile = "team"`, `${FULL}"profile" = "team"`, + `${FULL}[[profiles]`, `${FULL}[profiles]]`, `${FULL}[features]\nnot_valid = [`, + `${FULL}[features]\nx = "unterminated`, `${FULL}[features]\nx = 1 2`, + `${FULL}[features]\nx = [1 2]`, `${FULL}[features]\nx = {a = 1 b = 2}`, + ]) { + writeFileSync(path, value); + assert.equal(f.send(), "", value); + } + writeFileSync(path, `${FULL}[features]\nenabled = true\n[sandbox_workspace_write]\nwritable_roots = [\n "/tmp",\n]\n`); + assert.equal(f.send(), ALLOW); + writeFileSync(path, `${FULL}[features]\nsettings = { enabled = true, retries = 2 }\nnotes = """hello\nworld"""\nraw = '''one\ntwo'''\n`); + assert.equal(f.send(), ALLOW, "closed compound values"); + writeFileSync(path, `${FULL}${"#".repeat(1024 * 1024)}`); + assert.equal(f.send(), "", "oversized Codex config"); +}); + +test("ignores project-local opt-in and noncovered tool", (t) => { + const f = fixture(t); + writeFileSync(join(f.clawHome, "config.json"), "{}"); + writeFileSync(join(f.cwd, "codexclaw.json"), '{"permissions":{"agentCreatedThreadAutoAllow":true}}'); + mkdirSync(join(f.cwd, ".codexclaw")); + writeFileSync(join(f.cwd, ".codexclaw", "config.json"), '{"permissions":{"agentCreatedThreadAutoAllow":true}}'); + assert.equal(f.send(), ""); + assert.equal(f.send({}, { ...f.env, CODEXCLAW_HOME: join(f.cwd, ".codexclaw") }), ""); + assert.equal(f.send({}, { ...f.env, CODEXCLAW_HOME: "relative" }), ""); + assert.equal(f.send({}, { ...f.env, CODEX_HOME: "relative" }), ""); + writeFileSync(join(f.clawHome, "config.json"), '{"permissions":{"agentCreatedThreadAutoAllow":true}}'); + for (const tool_name of ["Edit", "mcp_tool", "functions.exec", "", "mcp__", "mcp__server", "mcp____tool", "mcp__server__"]) { + assert.equal(f.send({ tool_name }), "", tool_name); + } + assert.equal(f.send({ tool_name: "mcp__codex_app__create_thread" }), ALLOW); +}); + +test("global opt-in rejects symlinked project config and accepts plain outside config", (t) => { + const f = fixture(t); + const projectConfig = join(f.cwd, ".codexclaw", "config.json"); + mkdirSync(dirname(projectConfig)); + writeFileSync(projectConfig, '{"permissions":{"agentCreatedThreadAutoAllow":true}}'); + const linkedHome = join(f.root, "linked-home"); + symlinkSync(dirname(projectConfig), linkedHome); + assert.equal(f.send({}, { ...f.env, CODEXCLAW_HOME: linkedHome }), ""); + const outsideConfig = join(f.clawHome, "config.json"); + rmSync(outsideConfig); + symlinkSync(projectConfig, outsideConfig); + assert.equal(f.send(), ""); + rmSync(outsideConfig); + writeFileSync(outsideConfig, '{"permissions":{"agentCreatedThreadAutoAllow":true}}'); + assert.equal(f.send(), ALLOW); +}); + +test("default home config remains eligible at home but a project symlink does not", (t) => { + const f = fixture(t); + const defaultDir = join(f.root, ".codexclaw"); + mkdirSync(defaultDir); + const defaultConfig = join(defaultDir, "config.json"); + writeFileSync(defaultConfig, '{"permissions":{"agentCreatedThreadAutoAllow":true}}'); + const env = { CODEX_HOME: f.codexHome, CODEXCLAW_HOME: "", HOME: f.root }; + const atHome = cli("permission-request", { ...f.input, cwd: f.root }, f.root, env); + assert.equal(atHome.status, 0, atHome.stderr); + assert.equal(atHome.stdout, ALLOW); + const projectConfig = join(f.cwd, "config.json"); + writeFileSync(projectConfig, '{"permissions":{"agentCreatedThreadAutoAllow":true}}'); + rmSync(defaultConfig); + symlinkSync(projectConfig, defaultConfig); + const project = cli("permission-request", f.input, f.cwd, env); + assert.equal(project.status, 0, project.stderr); + assert.equal(project.stdout, ""); +}); + +test("emits exact allow JSON and otherwise no stdout", (t) => { + const f = fixture(t); + assert.deepEqual(JSON.parse(f.send()), { hookSpecificOutput: { hookEventName: "PermissionRequest", decision: { behavior: "allow" } } }); + assert.equal(f.send(), ALLOW); + for (const raw of ["{", "", JSON.stringify({ ...f.input, hook_event_name: "Other" }), JSON.stringify({ ...f.input, hook_event_name: null })]) { + assert.equal(permission(raw, f.env), ""); + } + assert.equal(f.send({ transcript_path: join(f.root, "missing") }), ""); + writeFileSync(join(f.clawHome, "config.json"), `${" ".repeat(1024 * 1024)}{}`); + assert.equal(f.send(), "", "oversized global config"); +}); + +test("advises agent-created default thread without opt-in", (t) => { + const f = fixture(t); + writeFileSync(join(f.clawHome, "config.json"), "{}"); + const result = JSON.parse(f.advise()); + assert.match(result.systemMessage, /Full Access/); + assert.match(result.systemMessage, /agentCreatedThreadAutoAllow/); + assert.match(result.systemMessage, /one-time network/); + assert.equal(result.hookSpecificOutput.hookEventName, "SessionStart"); + assert.match(result.hookSpecificOutput.additionalContext, /network/); + assert.match(result.hookSpecificOutput.additionalContext, /git/); +}); + +test("advisory is silent outside degraded agent-created context", (t) => { + const f = fixture(t); + for (const value of ['approval_policy = "on-request"\nsandbox_mode = "danger-full-access"', 'approval_policy = "never"\nsandbox_mode = "workspace-write"', `${FULL}profile = "team"`]) { + writeFileSync(join(f.codexHome, "config.toml"), value); + assert.equal(f.advise(), ""); + } + writeFileSync(join(f.codexHome, "config.toml"), FULL); + for (const change of [{ permission_mode: "full-access" }, { agent_id: null }, { agent_type: "worker" }]) assert.equal(f.advise(change), ""); + writeFileSync(f.transcript, `${JSON.stringify({ type: "session_meta", payload: { id: "fixture-id", thread_source: "user" } })}\n`); + assert.equal(f.advise(), ""); + writeFileSync(f.transcript, "{bad\n"); + assert.equal(f.advise(), ""); +}); + +test("CLI hooks fail open before root-only subagent exit", (t) => { + const f = fixture(t); + for (const [verb, input] of [["permission-request", f.input], ["session-start-permission-advisory", { ...f.input, hook_event_name: "SessionStart" }]] as const) { + const positive = cli(verb, input, f.cwd, f.env); + assert.equal(positive.status, 0, positive.stderr); + assert.equal(verb === "permission-request" ? positive.stdout : JSON.parse(positive.stdout).hookSpecificOutput.hookEventName, verb === "permission-request" ? ALLOW : "SessionStart"); + for (const bad of ["{bad", "x".repeat(4 * 1024 * 1024 + 1)]) { + const result = cli(verb, bad, f.cwd, f.env); + assert.equal(result.status, 0, result.stderr); + assert.equal(result.stdout, ""); + } + } +}); + +test("CLI hooks stay active with PABCD disabled and never create project state", (t) => { + const f = fixture(t); + for (const mode of ["env", "project"] as const) { + const cwd = join(f.root, `project-${mode}`); + mkdirSync(cwd); + if (mode === "project") writeFileSync(join(cwd, "codexclaw.json"), '{"pabcd":{"enabled":false}}'); + const env = { ...f.env, CODEXCLAW_PABCD: mode === "env" ? "off" : "" }; + const input = { ...f.input, cwd }; + const allowed = cli("permission-request", input, cwd, env); + assert.equal(allowed.status, 0, allowed.stderr); + assert.equal(allowed.stdout, ALLOW); + const advised = cli("session-start-permission-advisory", { ...input, hook_event_name: "SessionStart" }, cwd, env); + assert.equal(advised.status, 0, advised.stderr); + assert.equal(JSON.parse(advised.stdout).hookSpecificOutput.hookEventName, "SessionStart"); + assert.equal(existsSync(join(cwd, ".codexclaw")), false); + writeFileSync(join(f.clawHome, "config.json"), "{}"); + assert.equal(cli("permission-request", input, cwd, env).stdout, ""); + writeFileSync(join(f.clawHome, "config.json"), '{"permissions":{"agentCreatedThreadAutoAllow":true}}'); + } +}); diff --git a/plugins/codexclaw/hooks/permission-request-allowing-agent-thread.json b/plugins/codexclaw/hooks/permission-request-allowing-agent-thread.json new file mode 100644 index 00000000..ffd41ef6 --- /dev/null +++ b/plugins/codexclaw/hooks/permission-request-allowing-agent-thread.json @@ -0,0 +1,13 @@ +{ + "hooks": { + "PermissionRequest": [{ + "matcher": "*", + "hooks": [{ + "type": "command", + "command": "node \"${PLUGIN_ROOT}/components/pabcd-state/dist/cli.js\" hook permission-request", + "timeout": 10, + "statusMessage": "(codexclaw) Checking agent-created thread permission" + }] + }] + } +} diff --git a/plugins/codexclaw/hooks/session-start-advising-agent-thread-permissions.json b/plugins/codexclaw/hooks/session-start-advising-agent-thread-permissions.json new file mode 100644 index 00000000..40d17b02 --- /dev/null +++ b/plugins/codexclaw/hooks/session-start-advising-agent-thread-permissions.json @@ -0,0 +1,12 @@ +{ + "hooks": { + "SessionStart": [{ + "hooks": [{ + "type": "command", + "command": "node \"${PLUGIN_ROOT}/components/pabcd-state/dist/cli.js\" hook session-start-permission-advisory", + "timeout": 10, + "statusMessage": "(codexclaw) Advising on agent-created thread permissions" + }] + }] + } +} From 8e4a3f898d87c6c01d272c3d3f2fc4e80600538c Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 01:49:08 +0900 Subject: [PATCH 40/90] docs(plan): fold wp3 audit round 2 (whole-file TOML check, canonical opt-in path, advice text) --- .../020_wp3_agent_thread_permissions.md | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md b/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md index 40bb4ffa..7cf4faa2 100644 --- a/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md +++ b/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md @@ -31,7 +31,7 @@ import { isAbsolute, join, relative, resolve } from "node:path"; const MAX_META_LINE_BYTES = 64 * 1024; const MAX_CONFIG_BYTES = 1024 * 1024; const ALLOW = '{"hookSpecificOutput":{"hookEventName":"PermissionRequest","decision":{"behavior":"allow"}}}'; -const MODEL_ADVICE = "This Codex Desktop agent-created thread may show approval prompts even though the user Codex config requests full access. Request escalation explicitly for network or git operations when needed; do not assume this hook changes sandbox or network access."; +const MODEL_ADVICE = "This Codex Desktop agent-created thread may show approval prompts even though the user Codex config requests full access. Request escalation explicitly for network or git operations when needed. If the user enabled permissions.agentCreatedThreadAutoAllow, codexclaw answers pending approvals, including one-time network requests, without prompting; it never changes this thread's sandbox."; const USER_ADVICE = "This agent-created thread started in the default approval mode despite your full-access Codex config, so approval prompts may appear. You can switch this thread to Full Access in the composer or enable permissions.agentCreatedThreadAutoAllow in your user-global Codexclaw config."; type JsonObject = Record; @@ -294,7 +294,7 @@ Cross-phase integration test (required, in `agent-thread-permissions.test.ts`): ## Out of scope -No upstream Codex patch, runtime permission-profile change, automatic global opt-in, repo-local permission setting, forked-thread allowance, trust-state forging, or network/sandbox widening. `021_wp3_dispatch_guidance.md` owns the separate dispatch and #265 checkpoint text. +No upstream Codex patch, runtime permission-profile change, automatic global opt-in, repo-local permission setting, forked-thread allowance, trust-state forging, or sandbox changes. One-time network-access approvals are answered under the opt-in (see the audit folds), which is disclosed in the opt-in documentation and the model advice. `021_wp3_dispatch_guidance.md` owns the separate dispatch and #265 checkpoint text. ## wp3 re-verification against codex/issue-train-wp2 (supersedes stale anchors above) @@ -314,3 +314,10 @@ Architect handle `01a0e3b8-436b-7203-a4f6-97d24b865814` re-checked this plan aft - **Malformed TOML fails closed.** Table headers must have paired delimiters (`[name]` or `[[name]]`); `[[profiles]`, `[profiles]]` and any other line starting with `[` return no decision. Tests: both malformed headers plus a valid `[features]` header after the two keys. - **`request_permissions` is out of scope.** The current host routes it straight to Guardian without PermissionRequest hooks (`codex-rs/core/src/session/mod.rs:2985-3008`, `approvals.rs:866-868`), so listing it would be a coverage claim no live path exercises. It is removed from `coveredTool` and the tests; revisit if the host starts routing it through hooks. - **A project cannot grant the opt-in.** The handler requires a string `cwd` in the hook input and ignores a `CODEXCLAW_HOME` that resolves inside that cwd. Test: absolute `CODEXCLAW_HOME=/.codexclaw` with `agentCreatedThreadAutoAllow: true` gets no decision; a sibling temp dir outside cwd with the same file allows. + + +## wp3 audit folds (round 2, supersede the reader code above where they differ) + +- **Whole-file TOML validation.** `codexConfigFullAccess` validates the entire file before trusting the two top-level keys; it no longer stops at the first header. A small structural scanner (no new dependency) walks the file: blank lines and `#` comments; table headers with paired delimiters; `key = value` statements where the value is a closed basic or literal string, a number, a boolean, an offset date-time, an inline table, or an array. Arrays and inline tables may span lines (the user's own config has `writable_roots = [` across lines); the scanner tracks bracket and brace depth outside strings and requires depth 0 at each statement end. Multi-line strings (`"""`, `'''`) are accepted when closed. Anything else, including an unclosed bracket at EOF, a stray token, or a duplicate top-level `approval_policy`/`sandbox_mode`, returns no decision. Tests: the user's pattern (keys, then `[features]`, then a multi-line `writable_roots` array) allows; `not_valid = [` at EOF after a valid table, `[[profiles]`, `[profiles]]`, `x = "unterminated` and `x = 1 2` get no decision. +- **Canonical containment for overrides only.** The default `~/.codexclaw/config.json` is user-global by definition and is always eligible, including for a thread whose cwd is the home directory. When `CODEXCLAW_HOME` is set, the handler resolves the real path of the config file it will read (`realpathSync`) and the real path of `cwd`, and gives no decision if the file lies inside the cwd tree. Tests: override `/.codexclaw` (no decision), override outside cwd that is a symlink into `/.codexclaw` (no decision), outside `config.json` symlinked to a project file (no decision), plain outside override (allow), default home with cwd = home (allow). +- **Advice and scope text.** `MODEL_ADVICE` and the out-of-scope line now say that the opt-in answers one-time network requests and never changes the sandbox. The opt-in documentation states the same. From 26155617a1e01cda33eda9a8908c6647eef56fd9 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 02:07:49 +0900 Subject: [PATCH 41/90] docs(pabcd): explain agent thread permission opt-in --- README.md | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/README.md b/README.md index bcf9d54e..d2b9d02c 100644 --- a/README.md +++ b/README.md @@ -74,6 +74,16 @@ Then restart Codex and approve the 24 hooks when prompted (upgrades ask again - "Interview me first, then draft a diff-level plan." - "Plan this with codexclaw PABCD and use multi-model subagents." +### Agent-created thread approvals (optional) + +Some Codex Desktop agent-created threads start with approval prompts even when your top-level Codex config requests `approval_policy = "never"` and `sandbox_mode = "danger-full-access"`. Codexclaw warns in those threads. To let its trusted PermissionRequest hook answer pending approvals, including one-time network-access requests, add this JSON boolean to `$CODEXCLAW_HOME/config.json` (default `~/.codexclaw/config.json`): + +```json +{"permissions":{"agentCreatedThreadAutoAllow":true}} +``` + +Preserve any other keys already in that file. The setting is off when absent and cannot be enabled by a project-local file. It applies only when the rollout identifies an agent-created root thread in the default approval mode and the user Codex config has both explicit top-level full-access values. It does not change the thread's sandbox or guarantee network or filesystem access; other hooks can still deny a request. New or changed hooks need Codex trust approval before they run. +

Update / uninstall / optional CLI From 4b9779aa1bc15a78b92acbd218c9e8103f37895c Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 01:50:03 +0900 Subject: [PATCH 42/90] docs(plan): fold wp3 audit round 3 (default path containment, user advice) --- .../260927_issue_train/020_wp3_agent_thread_permissions.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md b/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md index 7cf4faa2..b85a120e 100644 --- a/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md +++ b/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md @@ -32,7 +32,7 @@ const MAX_META_LINE_BYTES = 64 * 1024; const MAX_CONFIG_BYTES = 1024 * 1024; const ALLOW = '{"hookSpecificOutput":{"hookEventName":"PermissionRequest","decision":{"behavior":"allow"}}}'; const MODEL_ADVICE = "This Codex Desktop agent-created thread may show approval prompts even though the user Codex config requests full access. Request escalation explicitly for network or git operations when needed. If the user enabled permissions.agentCreatedThreadAutoAllow, codexclaw answers pending approvals, including one-time network requests, without prompting; it never changes this thread's sandbox."; -const USER_ADVICE = "This agent-created thread started in the default approval mode despite your full-access Codex config, so approval prompts may appear. You can switch this thread to Full Access in the composer or enable permissions.agentCreatedThreadAutoAllow in your user-global Codexclaw config."; +const USER_ADVICE = "This agent-created thread started in the default approval mode despite your full-access Codex config, so approval prompts may appear. You can switch this thread to Full Access in the composer, or set permissions.agentCreatedThreadAutoAllow to true in ~/.codexclaw/config.json so codexclaw answers these approvals for you, including one-time network requests."; type JsonObject = Record; @@ -319,5 +319,5 @@ Architect handle `01a0e3b8-436b-7203-a4f6-97d24b865814` re-checked this plan aft ## wp3 audit folds (round 2, supersede the reader code above where they differ) - **Whole-file TOML validation.** `codexConfigFullAccess` validates the entire file before trusting the two top-level keys; it no longer stops at the first header. A small structural scanner (no new dependency) walks the file: blank lines and `#` comments; table headers with paired delimiters; `key = value` statements where the value is a closed basic or literal string, a number, a boolean, an offset date-time, an inline table, or an array. Arrays and inline tables may span lines (the user's own config has `writable_roots = [` across lines); the scanner tracks bracket and brace depth outside strings and requires depth 0 at each statement end. Multi-line strings (`"""`, `'''`) are accepted when closed. Anything else, including an unclosed bracket at EOF, a stray token, or a duplicate top-level `approval_policy`/`sandbox_mode`, returns no decision. Tests: the user's pattern (keys, then `[features]`, then a multi-line `writable_roots` array) allows; `not_valid = [` at EOF after a valid table, `[[profiles]`, `[profiles]]`, `x = "unterminated` and `x = 1 2` get no decision. -- **Canonical containment for overrides only.** The default `~/.codexclaw/config.json` is user-global by definition and is always eligible, including for a thread whose cwd is the home directory. When `CODEXCLAW_HOME` is set, the handler resolves the real path of the config file it will read (`realpathSync`) and the real path of `cwd`, and gives no decision if the file lies inside the cwd tree. Tests: override `/.codexclaw` (no decision), override outside cwd that is a symlink into `/.codexclaw` (no decision), outside `config.json` symlinked to a project file (no decision), plain outside override (allow), default home with cwd = home (allow). +- **Canonical containment of the file actually read.** For both the default `~/.codexclaw/config.json` and a `CODEXCLAW_HOME` override, the handler resolves the real path of the config file it will read (`realpathSync`) and gives no decision when that real path lies inside the real cwd tree, with one exception: the unresolved default path itself (`/.codexclaw/config.json` when it is a regular file, not a symlink) stays eligible when cwd is the home directory, because that file is the user-global config by definition. A symlinked default file is judged by its target. Tests: override `/.codexclaw` (no decision), override outside cwd that is a symlink into `/.codexclaw` (no decision), outside `config.json` symlinked to a project file (no decision), plain outside override (allow), default home with cwd = home and a regular file (allow), default `~/.codexclaw/config.json` symlinked into the project (no decision). A test asserts `USER_ADVICE` mentions `agentCreatedThreadAutoAllow` and one-time network requests. - **Advice and scope text.** `MODEL_ADVICE` and the out-of-scope line now say that the opt-in answers one-time network requests and never changes the sandbox. The opt-in documentation states the same. From 08683b65d44116feae37022c49ef7689a8603a2e Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 01:57:29 +0900 Subject: [PATCH 43/90] docs(pabcd): clarify bounded worktree worker dispatch --- .../skills/pabcd/references/delegation.md | 47 +++++-- .../pabcd/references/dispatch-surfaces.md | 121 +++++++++++------- 2 files changed, 110 insertions(+), 58 deletions(-) diff --git a/plugins/codexclaw/skills/pabcd/references/delegation.md b/plugins/codexclaw/skills/pabcd/references/delegation.md index 848b4553..c0fad64e 100644 --- a/plugins/codexclaw/skills/pabcd/references/delegation.md +++ b/plugins/codexclaw/skills/pabcd/references/delegation.md @@ -3,7 +3,9 @@ This file assumes the surface is already chosen and describes the **subagent** packet. [Dispatch surfaces](dispatch-surfaces.md) owns the choice between a subagent and a separate Codex task, and the fact that a subagent runs in this -session's own working directory rather than a copy of it. +session's native working directory rather than a copy of it. A bounded worker +can operate in a separately created managed worktree only when its packet +supplies that absolute path and it uses it as every shell command's workdir. The main session owns the plan, host goal, and every PABCD transition. At P, consult a read-only architect; at A, dispatch an independent reviewer. @@ -20,6 +22,26 @@ Subagents return evidence and unresolved judgments; the main session decides and integrates. Dispatch only specifiable work whose coordination cost is justified (DISPATCH-ECONOMY-01). +### Optional worker progress checkpoint (#265) + +For a long bounded write packet, the coordinator may grant a specific `PROGRESS.md` +path inside the worker's assigned worktree. The worker may update it after a +coherent edit or check with three fields: `Done`, `Remaining`, and `Partial files` +(absolute paths plus what is incomplete). Example: + +```text +Done: parsed hook input and added the first regression test +Remaining: add manifest entries; run focused tests +Partial files: /absolute/worktree/path/src/agent-thread-permissions.ts — parser branch incomplete +``` + +The checkpoint is a handoff hint, not completion proof or a new source of +authority. On interruption, the coordinator checks that the first worker has +stopped, reads `PROGRESS.md` and the named files, then gives the replacement +worker the same bounded packet, worktree path, and remaining work. The +replacement verifies the actual file state before editing. Without a granted +path, the worker does not create `PROGRESS.md`. + **DISPATCH-PROMOTE-01 (DEFAULT).** After checking a child's evidence, main records a short synthesis of what it accepted: the reusable result, the failure cause or procedure worth keeping, its provenance, and any claim still unresolved. This is not @@ -224,17 +246,24 @@ names in your own session: `worktree` is what gives a lane its own checkout. Creating a thread is user-visible; messaging one is not commanding it. +A `create_thread` child may start with reduced approval permission even when +the coordinator is full-access; this also occurs for projectless targets. Check +the child's actual permission mode before assigning unattended writes. A +bounded checkout worker can instead use `create_worktree` plus a subagent with +the returned absolute path as every shell workdir. This does not give the +subagent its own task, goal or PABCD state. **Delegation safeguards:** -- **DISPATCH-ISOLATION-01:** subagent lanes are not isolated environments — they - all run in this session's working directory, so "isolation" here means scope - discipline, not separation. Give every lane explicit read and write access lists - with no overlap, and never share in-progress output across lanes. Concurrent - lanes must never run branch-level git operations (`checkout`, `switch`, - `branch`, `stash`, `reset`, `rebase`, `merge`, `pull`): those act on one shared - HEAD and a per-file write scope does not make them safe. Work that genuinely - needs its own branch or checkout is thread work, not a subagent lane. +- **DISPATCH-ISOLATION-01:** subagents inherit the parent's native cwd; they + do not get a copied checkout. Give concurrent workers disjoint read/write + scopes. For a bounded worker in a managed worktree, assign one absolute + worktree path and require the shell workdir on every command; use absolute + file paths for edits. Different workers get different worktrees. Never run + concurrent branch-level operations (`checkout`, `switch`, `branch`, `stash`, + `reset`, `rebase`, `merge`, `pull`) in one checkout; a file scope cannot + separate one HEAD. Work that needs its own goal or PABCD cycle stays a + separate thread task. - **REVIEW-DECORRELATE-01:** prefer an independent context; use a different model family only when host policy and user authorization permit the override. Otherwise inherit and record that family-level independence was not established. diff --git a/plugins/codexclaw/skills/pabcd/references/dispatch-surfaces.md b/plugins/codexclaw/skills/pabcd/references/dispatch-surfaces.md index a638a7d0..c64702c8 100644 --- a/plugins/codexclaw/skills/pabcd/references/dispatch-surfaces.md +++ b/plugins/codexclaw/skills/pabcd/references/dispatch-surfaces.md @@ -10,8 +10,8 @@ the choice is made, and its V1/V2 section owns the tool schemas. different mechanisms answer to those words and they are not substitutes: - A **subagent** is a leaf spawned with the collab tools (`spawn_agent` in the - `multi_agent_v1` or `collaboration` namespace). It runs **in the parent's own - working directory**. It has no session state, no host goal and no PABCD FSM. + `multi_agent_v1` or `collaboration` namespace). Its native cwd inherits the + parent's working directory. It has no session state, no host goal and no PABCD FSM. - A **thread** is a separate Codex task created with the desktop task tools (`create_thread` and its family). With `environment: worktree` it gets its own checkout; with `environment: local` it shares the project checkout. Either way @@ -20,11 +20,15 @@ different mechanisms answer to those words and they are not substitutes: Isolation comes from the environment, not from being a task. A `local` thread is an independent owner sharing one checkout; a `worktree` thread is an independent -owner with its own. Lane work needs the second. +owner with its own. An independent task lane needs the second. -A **lane** is thread work; a **worker inside a lane** is subagent work. N lanes -means N worktree threads, and the workers inside each lane are that lane's -subagents — they cannot collide across lanes because the worktrees differ. +An **independent task lane** owns a goal, PABCD cycle or long-running branch/CI +lifecycle: use one worktree thread per lane. A **bounded checkout worker** needs +only a disjoint checkout and returns a patch or evidence to the coordinator: +create a managed worktree, then give its absolute path to a subagent. The +subagent's native cwd still inherits the coordinator's; the packet must require +that path as the shell workdir on every command. Workers do not acquire their +own goal or PABCD state. Different workers must use different worktrees. Say which one you are creating, in those words, before you create it. @@ -32,9 +36,9 @@ Say which one you are creating, in those words, before you create it. | | Subagent (`spawn_agent`) | Thread (`create_thread`) | |---|---|---| -| Working directory | the parent's, unchanged; never a copy | its own, with `environment: worktree`; the shared project checkout with `local` | -| Git branch and HEAD | the parent's | its own under `worktree`; shared under `local` | -| Edits visible to the parent | immediately, as the parent's own uncommitted changes | only through git | +| Working directory | native cwd inherits the parent's; a bounded worker must pass its assigned managed-worktree path as the shell workdir on every command | its own, with `environment: worktree`; the shared project checkout with `local` | +| Git branch and HEAD | native cwd points at the parent's; commands run in an assigned managed worktree see that worktree's branch and HEAD | its own under `worktree`; shared under `local` | +| Edits visible to the parent | immediately in the selected checkout; a managed worktree has its own branch and files | only through git | | Thread id | yes, its own | yes, its own | | `.codexclaw` session state | none | its own | | Host goal | none; it must not call `create_goal` | its own, keyed to the task | @@ -48,7 +52,7 @@ Say which one you are creating, in those words, before you create it. A distinct thread id is the trap. A subagent has one, which is why "thread" feels like the right word for it. It proves nothing about the filesystem. -## DISPATCH-SHARED-TREE-01 (STRICT) — subagents share your checkout +## DISPATCH-SHARED-TREE-01 (STRICT) — subagents inherit your cwd A spawned child inherits the parent's cwd. Measured on 2026-09-13: a probe subagent reported the parent's `pwd`, the parent's `git rev-parse --show-toplevel`, @@ -59,46 +63,60 @@ paths; no worktree is created anywhere on that path. Therefore: -- Write scopes across concurrent subagents must not overlap. -- **Never** run two subagents that perform branch-level git operations at the - same time. `checkout`, `switch`, `branch`, `stash`, `reset`, `rebase`, `merge` - and `pull` act on one shared HEAD; two children doing that corrupt each other's - work regardless of how their file scopes were divided. A per-file write scope - does not make concurrent branch work safe. -- Tell the child it shares your tree. It cannot infer this: on V1 the host tool +- Write scopes across concurrent subagents must not overlap. A coordinator + assigning separate managed worktrees must give each worker a different path. +- **Never** run concurrent branch-level git operations in one checkout. + `checkout`, `switch`, `branch`, `stash`, `reset`, `rebase`, `merge` and `pull` + change that checkout's HEAD or index; a per-file write scope does not separate + them. Operations in different worktrees do not share one HEAD, but each branch + still needs one owner and an explicit integration order. +- Tell the child its native cwd is your tree. It cannot infer this: on V1 the host tool description says the opposite, instructing the caller to have the child "edit files directly in its forked workspace". There is no forked workspace. `fork_context` and `fork_turns` fork conversation history, not the filesystem. Only the V2 usage hint states the shared directory, so a V1 session is never told it by the runtime. +- For a managed-worktree worker, instruct the subagent to pass the absolute + worktree path as the shell tool's workdir on **every** command, including + `git status`, tests and reads. Use absolute paths for file edits. Its native + cwd and relative-path defaults do not move when the worktree is created. ## DISPATCH-ROUTE-01 (STRICT) — routing the work Route by what the work needs to own, not by how parallel it is: -- Needs its own branch, checkout, or long-running merge/CI lane -> **thread**, - one per lane, created with `environment: worktree`. A `local` thread does not - give the lane a checkout of its own. -- Needs its own goal or its own PABCD cycle -> **thread**. -- Is a bounded slice inside a lane that already owns its checkout -> **subagent** - of that lane's thread. -- Is a bounded slice of the tree you are already editing, returning evidence or a - patch rather than owning a branch -> **subagent**. -- Is read-only research -> **subagent**, by default. It cannot collide because it - writes nothing, which is also why read-only fan-out is not a template for - parallel write work. - -"Merge these lanes in parallel", "prepare N stacks at once", "run these branches -concurrently" are thread work. Spawning N subagents for N branches puts N writers -on one HEAD. +- Needs its own goal, PABCD cycle, user-visible task, or long-running + merge/CI lifecycle -> **thread**, one per independent task lane, with + `environment: worktree` for an isolated checkout. A `local` thread shares + the checkout. +- Needs an isolated checkout for a bounded, coordinator-owned write packet + while the coordinator is full-access -> call `create_worktree`, wait for its + completed absolute workspace path, then spawn a **subagent** with that path + and an instruction to pass it as the shell workdir on every command. Give + concurrent workers disjoint worktrees and prohibit concurrent branch + operations in one checkout. The coordinator owns goal/PABCD and integration. +- Is a bounded slice inside a thread lane that already owns its checkout -> + **subagent** of that thread. +- Is a bounded slice of the checkout you are already editing, returning + evidence or a patch -> **subagent** with disjoint file scope. +- Is read-only research -> **subagent**, by default. Read-only fan-out is not + a template for parallel writes. + +`create_thread` children may start with reduced approval permission, including +projectless targets. Confirm their actual permission state before planning an +unattended write lane. The bounded worktree/subagent route does not grant new +permissions; it uses the coordinator's inherited subagent permission and an +explicit checkout path. When a lane needs independent goal/PABCD ownership, +keep the thread route and handle its actual permission state. ## DISPATCH-AUTHORITY-01 — asking for lane work is asking for the lanes Creating a thread is user-visible, so it needs a user request. A request for -parallel branch or worktree lanes **is** that request: the lanes are the -mechanism the work needs, not a separate deliverable the user forgot to ask for. -Do not read the general "create a task only when the user explicitly asks" rule -as a reason to downgrade lane work onto the shared tree — that trades a visible +independent task lanes **is** that request: the lanes are the mechanism the work +needs, not a separate deliverable the user forgot to ask for. Bounded checkout +workers stay under the coordinator and follow DISPATCH-ROUTE-01. Do not read +the general "create a task only when the user explicitly asks" rule as a reason +to put independent task lanes onto the shared tree — that trades a visible question for a silent collision. Where the shape is genuinely unclear, ask once and name what you would create @@ -107,9 +125,11 @@ do not treat silence as a refusal of the surface the work requires. ## Parallel lanes, and the shape that works -N independent lanes means N `worktree` threads, N checkouts, N FSMs. The parent -coordinates with `wait_threads` and integrates; it does not advance any child's -FSM, and a child does not advance the parent's. +N independent task lanes mean N `worktree` threads, N checkouts and N FSMs. +The coordinator uses `wait_threads` and integrates; neither side advances the +other's FSM. N bounded checkout workers mean N managed worktrees and N +subagents, with one coordinator goal/FSM. The coordinator uses the returned +subagent handles and checks each worktree's files before integration. ### Record the lane before you need it (DISPATCH-LANE-ID-01, DEFAULT) @@ -181,9 +201,10 @@ fails outright with `agent thread limit reached`, at six per session by default (`agents.max_threads`; on V2 `max_concurrent_threads_per_session` minus one for the session itself). -So cross-branch fan-out belongs to lanes, and concurrency inside one lane's tree belongs -to that lane's subagents — run them in waves, say the wave size, and close finished -agents, because a completed agent holds its slot until it is closed. "Unlimited parallel +Independent task fan-out belongs to thread lanes. Bounded checkout workers are +subagents even in separate worktrees, so they share the session's subagent cap. +Run them in waves, say the wave size, and close finished agents, because a +completed agent holds its slot until it is closed. "Unlimited parallel subagents" is not a shape the host offers. Lanes and the parent watching them usually draw on the same credentials and the same @@ -198,19 +219,21 @@ reserve headroom for the workers. Back off on evidenced limit responses. Do not every 403 is exhaustion, that every account has the same allowance, or that rate-limit categories are interchangeable. This is guidance for the coordinator, not a limiter. -Threads and subagents then compose. A lane thread spawns its own subagents inside -its own worktree, and subagents belonging to different lanes cannot collide -**because those worktrees differ** — not because their parents are different -tasks. Two `local` threads on one checkout collide exactly like two subagents do. -The shape that scales is worktrees for isolation and subagents for concurrency -within an isolated tree. +Threads and subagents compose in independent task lanes: each worktree thread +spawns bounded subagents inside its checkout. A full-access coordinator can +also assign separate managed worktrees directly to bounded subagents. In both +forms, different worktrees provide file and HEAD isolation; different thread +ids do not. Two `local` threads on one checkout still collide. ### The lane manifest (DISPATCH-LANE-MANIFEST-01, DEFAULT) -Lanes are independent tasks, so nothing in the system knows two of them were handed the +Independent task lanes are separate tasks, so nothing in the system knows two were handed the same issue until their pull requests collide. One shared record makes that visible before the branches diverge. Per lane: repository, lane id, task and host id, worktree, branch, base ref and sha, head sha, issue, owner, scope and status. +A bounded worktree worker remains under its coordinator and does not invent a +thread id or its own FSM. Record its worktree path and assigned scope in the +coordinator's packet or progress record instead. ```json { From 384d699fc621868edd648893115f24c29f420b02 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 02:08:52 +0900 Subject: [PATCH 44/90] build: regenerate dist for agent-thread permissions --- .../dist/agent-thread-permissions.js | 228 ++++++++++++++++++ .../components/pabcd-state/dist/cli.js | 17 ++ 2 files changed, 245 insertions(+) create mode 100644 plugins/codexclaw/components/pabcd-state/dist/agent-thread-permissions.js diff --git a/plugins/codexclaw/components/pabcd-state/dist/agent-thread-permissions.js b/plugins/codexclaw/components/pabcd-state/dist/agent-thread-permissions.js new file mode 100644 index 00000000..9827c3e2 --- /dev/null +++ b/plugins/codexclaw/components/pabcd-state/dist/agent-thread-permissions.js @@ -0,0 +1,228 @@ +import { closeSync, lstatSync, openSync, readFileSync, readSync, realpathSync, statSync } from "node:fs"; +import { homedir } from "node:os"; +import { isAbsolute, join, relative, resolve } from "node:path"; + +const MAX_META_LINE_BYTES = 64 * 1024; +const MAX_CONFIG_BYTES = 1024 * 1024; +const ALLOW = '{"hookSpecificOutput":{"hookEventName":"PermissionRequest","decision":{"behavior":"allow"}}}'; +const MODEL_ADVICE = "This Codex Desktop agent-created thread may show approval prompts even though the user Codex config requests full access. Request escalation explicitly for network or git operations when needed. If the user enabled permissions.agentCreatedThreadAutoAllow, codexclaw answers pending approvals, including one-time network requests, without prompting; it never changes this thread's sandbox."; +const USER_ADVICE = "This agent-created thread started in the default approval mode despite your full-access Codex config, so approval prompts may appear. You can switch this thread to Full Access in the composer, or set permissions.agentCreatedThreadAutoAllow to true in ~/.codexclaw/config.json so codexclaw answers these approvals for you, including one-time network requests."; + + + +function object(value ) { + return value !== null && typeof value === "object" && !Array.isArray(value) + ? value : null; +} + +function readFirstRecord(path ) { + if (!isAbsolute(path)) return null; + const fd = openSync(path, "r"); + try { + const buffer = Buffer.alloc(MAX_META_LINE_BYTES + 1); + let used = 0; + while (used < buffer.length) { + const count = readSync(fd, buffer, used, buffer.length - used, used); + if (count === 0) break; + used += count; + const end = buffer.subarray(0, used).indexOf(10); + if (end >= 0) return end > MAX_META_LINE_BYTES ? null : + object(JSON.parse(buffer.subarray(0, end).toString("utf8"))); + } + if (used === 0 || used > MAX_META_LINE_BYTES) return null; + return object(JSON.parse(buffer.subarray(0, used).toString("utf8"))); + } finally { + closeSync(fd); + } +} + +function agentCreatedRoot(input ) { + if (input.permission_mode !== "default" || + typeof input.session_id !== "string" || input.session_id.length === 0 || + typeof input.transcript_path !== "string" || + Object.hasOwn(input, "agent_id") || Object.hasOwn(input, "agent_type")) return false; + const record = readFirstRecord(input.transcript_path); + const payload = object(record?.payload); + return record?.type === "session_meta" && payload?.id === input.session_id && + payload.thread_source === "agent_created_thread" && payload.forked_from_id == null; +} + +function boundedText(path ) { + const stat = statSync(path); + if (!stat.isFile() || stat.size > MAX_CONFIG_BYTES) return null; + return readFileSync(path, "utf8"); +} + +function globalOptIn(env , cwd ) { + const override = env.CODEXCLAW_HOME?.trim(); + if (override && !isAbsolute(override)) return false; + const home = override || join(homedir(), ".codexclaw"); + const configPath = join(home, "config.json"); + const realCwd = realpathSync(cwd); + const realConfig = realpathSync(configPath); + const rel = relative(realCwd, realConfig); + const withinCwd = rel === "" || (!rel.startsWith("..") && !isAbsolute(rel)); + const defaultAtHome = !override && realCwd === realpathSync(homedir()) && + !lstatSync(configPath).isSymbolicLink() && lstatSync(configPath).isFile(); + if (withinCwd && !defaultAtHome) return false; + const config = object(JSON.parse(boundedText(realConfig) ?? "null")); + const permissions = object(config?.permissions); + return permissions?.agentCreatedThreadAutoAllow === true; +} + +/** A bounded structural scanner: unknown TOML is never permission evidence. */ +function validValue(source ) { + let index = 0; + const scalar = /^(?:true|false|[+-]?(?:0|[1-9](?:[0-9_]*[0-9])?)(?:\.[0-9_]+)?(?:[eE][+-]?[0-9_]+)?|[+-]?(?:inf|nan)|\d{4}-\d\d-\d\d(?:[Tt ]\d\d:\d\d:\d\d(?:\.\d+)?(?:[Zz]|[+-]\d\d:\d\d)?)?|\d\d:\d\d:\d\d(?:\.\d+)?|0[xX][0-9a-fA-F_]+|0[oO][0-7_]+|0[bB][01_]+)$/; + const skip = () => { + while (index < source.length) { + if (/\s/.test(source[index])) { index += 1; continue; } + if (source[index] === "#") { + const end = source.indexOf("\n", index); + index = end < 0 ? source.length : end; + continue; + } + break; + } + }; + const quoted = () => { + const quote = source[index]; + const triple = source.startsWith(quote.repeat(3), index); + const mark = triple ? quote.repeat(3) : quote; + index += mark.length; + while (index < source.length) { + if (source.startsWith(mark, index)) { index += mark.length; return true; } + if (!triple && /[\r\n]/.test(source[index])) return false; + if (quote === '"' && source[index] === "\\") index += 1; + index += 1; + } + return false; + }; + const key = () => { + if (source[index] === '"' || source[index] === "'") return quoted(); + const match = /^[A-Za-z0-9_-]+/.exec(source.slice(index)); + if (!match) return false; + index += match[0].length; + return true; + }; + const value = (depth ) => { + if (depth > 64) return false; + skip(); + const opener = source[index]; + if (opener === '"' || opener === "'") return quoted(); + if (opener === "[" || opener === "{") { + index += 1; + const closer = opener === "[" ? "]" : "}"; + skip(); + if (source[index] === closer) { index += 1; return true; } + while (index < source.length) { + if (opener === "{") { + if (!key()) return false; + skip(); + if (source[index++] !== "=") return false; + } + if (!value(depth + 1)) return false; + skip(); + if (source[index] === closer) { index += 1; return true; } + if (source[index++] !== ",") return false; + skip(); + if (opener === "[" && source[index] === closer) { index += 1; return true; } + } + return false; + } + const bare = /^[^\s,\]\}#]+/.exec(source.slice(index)); + if (!bare || !scalar.test(bare[0])) return false; + index += bare[0].length; + return true; + }; + if (!value(0)) return false; + skip(); + return index === source.length; +} + +function validTomlAndTopLevel(content ) { + const seen = new Map (); + let inTopLevel = true; + const lines = content.replace(/^\uFEFF/, "").split(/\r?\n/); + for (let index = 0; index < lines.length; index += 1) { + const line = lines[index].trim(); + if (line === "" || line.startsWith("#")) continue; + if (line.startsWith("[")) { + if (!/^(?:\[\[[^[\]\r\n]+\]\]|\[[^[\]\r\n]+\])\s*(?:#.*)?$/.test(line)) return false; + inTopLevel = false; + continue; + } + const assignment = /^([A-Za-z0-9_-]+|"[^"\r\n]+"|'[^'\r\n]+')\s*=\s*(.*)$/.exec(line); + if (!assignment) return false; + const key = assignment[1].replace(/^["']|["']$/g, ""); + if (inTopLevel && key === "profile") return false; + let value = assignment[2]; + while (!validValue(value)) { + // Only arrays, inline tables and triple strings may continue onto another line. + if (!/^(?:\[|\{|"""|''')/.test(value) || index + 1 >= lines.length) return false; + value += `\n${lines[++index]}`; + } + if (inTopLevel && (key === "approval_policy" || key === "sandbox_mode")) { + if (seen.has(key)) return false; + const exact = /^"([^"\r\n]*)"\s*(?:#.*)?$/.exec(value); + if (!exact) return false; + seen.set(key, exact[1]); + } + } + return seen.get("approval_policy") === "never" && + seen.get("sandbox_mode") === "danger-full-access"; +} + +function codexConfigFullAccess(env ) { + const override = env.CODEX_HOME?.trim(); + if (override && !isAbsolute(override)) return false; + const home = override || join(homedir(), ".codex"); + const content = boundedText(join(home, "config.toml")); + if (content === null) return false; + return validTomlAndTopLevel(content); +} + +function coveredTool(name ) { + return typeof name === "string" && + (["Bash", "write_stdin", "apply_patch"].includes(name) || + isMcpToolName(name)); +} + +/** `mcp____` with both segments present. */ +function isMcpToolName(name ) { + const match = /^mcp__(.+?)__(.+)$/.exec(name); + return !!match && match[1].replace(/_/g, "") !== "" && match[2].replace(/_/g, "") !== ""; +} + +function parseHook(raw , event ) { + const input = object(JSON.parse(raw)); + return input?.hook_event_name === event ? input : null; +} + +export function handleAgentThreadPermissionRequest( + raw , env = process.env, +) { + try { + const input = parseHook(raw, "PermissionRequest"); + return input && typeof input.cwd === "string" && input.cwd !== "" && + coveredTool(input.tool_name) && globalOptIn(env, input.cwd) && + agentCreatedRoot(input) && codexConfigFullAccess(env) ? ALLOW : ""; + } catch { + return ""; + } +} + +export function handleAgentThreadSessionStartAdvisory( + raw , env = process.env, +) { + try { + const input = parseHook(raw, "SessionStart"); + if (!input || !agentCreatedRoot(input) || !codexConfigFullAccess(env)) return ""; + return `${JSON.stringify({ + systemMessage: USER_ADVICE, + hookSpecificOutput: { hookEventName: "SessionStart", additionalContext: MODEL_ADVICE }, + })}\n`; + } catch { + return ""; + } +} diff --git a/plugins/codexclaw/components/pabcd-state/dist/cli.js b/plugins/codexclaw/components/pabcd-state/dist/cli.js index 79eab444..4df4207f 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/cli.js +++ b/plugins/codexclaw/components/pabcd-state/dist/cli.js @@ -340,11 +340,28 @@ async function main() { const stdin = readStdin(); if (stdin.overflow) { + if (event === "permission-request" || event === "session-start-permission-advisory") { + process.exit(0); + } const denied = oversizedHookOutput(event); if (denied) process.stdout.write(denied); process.exit(denied ? 0 : 1); } const raw = stdin.raw; + if (event === "permission-request" || event === "session-start-permission-advisory") { + try { + recordHookInvocation(raw, "pabcd-state", event, import.meta.url); + const { handleAgentThreadPermissionRequest, handleAgentThreadSessionStartAdvisory } = + await import("./agent-thread-permissions.js"); + const result = event === "permission-request" + ? handleAgentThreadPermissionRequest(raw) + : handleAgentThreadSessionStartAdvisory(raw); + if (result) process.stdout.write(result); + } catch { + // Fail open: no decision/advisory, exit 0. + } + process.exit(0); + } recordHookInvocation(raw, "pabcd-state", event, import.meta.url); let output = ""; From 2c07284adcc02090f4bdadaea41211a72de7ea59 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 02:13:44 +0900 Subject: [PATCH 45/90] wp3: hook counts, inventory 3667, changelog --- CHANGELOG.md | 2 ++ README.ko.md | 8 ++++---- README.md | 8 ++++---- README.zh.md | 8 ++++---- .../components/cxc-ops/test/manifest-targets.test.ts | 7 ++++--- plugins/codexclaw/inventory.json | 12 ++++++++++++ plugins/codexclaw/test/hook-e2e.test.mjs | 3 ++- 7 files changed, 32 insertions(+), 16 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 117e214a..988480d9 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,6 +8,8 @@ All notable changes to codexclaw are documented here. The format follows ### Added +- Codex Desktop sometimes starts threads created by `create_thread` with on-request approvals even when the user's config is full access (openai/codex #33282). A SessionStart advisory now tells the model and the user when an agent-created thread starts that way. An opt-in PermissionRequest hook (`permissions.agentCreatedThreadAutoAllow: true` in `~/.codexclaw/config.json`, off by default) answers those threads' approval prompts, including one-time network requests, only when the user's top-level `config.toml` sets `approval_policy = "never"` and `sandbox_mode = "danger-full-access"`; it never changes the thread's sandbox, never denies, and ignores project-local config. Two new hooks (31 total) need trust approval after upgrade. +- Dispatch guidance: for bounded worktree lanes a full-access coordinator can create a managed worktree and hand a subagent that path as its shell `workdir`, which keeps the coordinator's permission; workers may keep an optional `PROGRESS.md` checkpoint so a replacement can resume from files (#265, guidance only). - `CODEXCLAW_PABCD=off` (or `on`) and project `codexclaw.json` `{"pabcd": {"enabled": false}}` turn the PABCD hook policy off while keeping the worktree, memory-write, automation-ownership and apply_patch lint guards and recall active. A recognized environment value wins over the project file in both directions (#252). - When codexclaw creates a project's `.codexclaw` folder, it also writes `.codexclaw/.gitignore` so session state, ledgers and evidence stay out of git; user-authored `rules/*.md` stay committable unless an ancestor ignore rule hides the folder. Existing `.codexclaw` folders are never modified. Lazy creation of session state is deferred (#255, partial). diff --git a/README.ko.md b/README.ko.md index 2e32567a..9fbd0870 100644 --- a/README.ko.md +++ b/README.ko.md @@ -13,9 +13,9 @@

CI - 3,650 tests + 3,667 tests 29 skills - 29 hooks + 31 hooks Documentation MIT

@@ -60,7 +60,7 @@ codex plugin marketplace add https://github.com/lidge-jun/codexclaw codex plugin add codexclaw@codexclaw ``` -설치 후 Codex를 재시작하고 뜨는 승인 창에서 24개 훅을 승인하면 된다(업그레이드 후에도 다시 승인 — 콘텐츠 해시 신뢰 모델). 채팅에서 바로 쓸 수 있고, 터미널 표면도 같이 배송된다 — 페이로드에 자체 `cxc` 디스패처가 들어 있어 에이전트의 `cxc orchestrate` 명령이 모든 설치에서 동작한다: +설치 후 Codex를 재시작하고 뜨는 승인 창에서 31개 훅을 승인하면 된다(업그레이드 후에도 다시 승인 — 콘텐츠 해시 신뢰 모델). 채팅에서 바로 쓸 수 있고, 터미널 표면도 같이 배송된다 — 페이로드에 자체 `cxc` 디스패처가 들어 있어 에이전트의 `cxc orchestrate` 명령이 모든 설치에서 동작한다: - `orchestrate status` — PABCD 상태 머신 확인 - "Interview me first, then draft a diff-level plan." @@ -176,7 +176,7 @@ plugins/codexclaw/ │ ├── recall/ past-session + memory store search │ └── repo-map/ tree-sitter + PageRank structure map │ -├── hooks/ 24 active hooks across the session lifecycle +├── hooks/ 31 active hooks across the session lifecycle │ ├── session-start-* provider bridge, PABCD bootstrap, map affordance, recall context │ ├── user-prompt-submit-* PABCD trigger detection, recall intent │ ├── pre-tool-use-* skill attach, goal guards, patch lint, interview guard diff --git a/README.md b/README.md index d2b9d02c..22720140 100644 --- a/README.md +++ b/README.md @@ -13,9 +13,9 @@

CI - 3,650 tests + 3,667 tests 29 skills - 29 hooks + 31 hooks Documentation MIT

@@ -68,7 +68,7 @@ codex plugin marketplace add https://github.com/lidge-jun/codexclaw codex plugin add codexclaw@codexclaw ``` -Then restart Codex and approve the 24 hooks when prompted (upgrades ask again — content-hash trust). Everything runs from chat, and the terminal surface ships too — the payload includes its own `cxc` dispatcher, so agent-driven `cxc orchestrate` commands work on every install: +Then restart Codex and approve the 31 hooks when prompted (upgrades ask again — content-hash trust). Everything runs from chat, and the terminal surface ships too — the payload includes its own `cxc` dispatcher, so agent-driven `cxc orchestrate` commands work on every install: - `orchestrate status` — check the PABCD state machine - "Interview me first, then draft a diff-level plan." @@ -196,7 +196,7 @@ plugins/codexclaw/ │ ├── recall/ past-session + memory store search │ └── repo-map/ tree-sitter + PageRank structure map │ -├── hooks/ 24 active hooks across the session lifecycle +├── hooks/ 31 active hooks across the session lifecycle │ ├── session-start-* provider bridge, PABCD bootstrap, map affordance, recall context │ ├── user-prompt-submit-* PABCD trigger detection, recall intent │ ├── pre-tool-use-* skill attach, goal guards, patch lint, interview guard diff --git a/README.zh.md b/README.zh.md index d78a7178..a2020ae8 100644 --- a/README.zh.md +++ b/README.zh.md @@ -13,9 +13,9 @@

CI - 3,650 tests + 3,667 tests 29 skills - 29 hooks + 31 hooks Documentation MIT

@@ -60,7 +60,7 @@ codex plugin marketplace add https://github.com/lidge-jun/codexclaw codex plugin add codexclaw@codexclaw ``` -然后重启 Codex,并在弹出的审批中批准 24 个 hooks(升级后需再次批准——内容哈希信任模型)。既可以直接在聊天中使用,终端界面也随包提供——payload 自带 `cxc` 调度器,代理的 `cxc orchestrate` 命令在任何安装方式下都能运行: +然后重启 Codex,并在弹出的审批中批准 31 个 hooks(升级后需再次批准——内容哈希信任模型)。既可以直接在聊天中使用,终端界面也随包提供——payload 自带 `cxc` 调度器,代理的 `cxc orchestrate` 命令在任何安装方式下都能运行: - `orchestrate status` — 查看 PABCD 状态机 - "Interview me first, then draft a diff-level plan." @@ -175,7 +175,7 @@ plugins/codexclaw/ │ ├── recall/ past-session + memory store search │ └── repo-map/ tree-sitter + PageRank structure map │ -├── hooks/ 24 active hooks across the session lifecycle +├── hooks/ 31 active hooks across the session lifecycle │ ├── session-start-* provider bridge, PABCD bootstrap, map affordance, recall context │ ├── user-prompt-submit-* PABCD trigger detection, recall intent │ ├── pre-tool-use-* skill attach, goal guards, patch lint, interview guard diff --git a/plugins/codexclaw/components/cxc-ops/test/manifest-targets.test.ts b/plugins/codexclaw/components/cxc-ops/test/manifest-targets.test.ts index 256faf87..11ccc0ce 100644 --- a/plugins/codexclaw/components/cxc-ops/test/manifest-targets.test.ts +++ b/plugins/codexclaw/components/cxc-ops/test/manifest-targets.test.ts @@ -241,9 +241,10 @@ test("C2b: deleting a shared target yields one issue per reference", () => { rmSync(join(root, "components/pabcd-state/dist/cli.js")); // 17 is hardcoded: seventeen hooks reference pabcd-state/dist/cli.js today (11 + // the 3 worktree-guard hooks 260804 + the review observer 260815 + the - // memory-write gate 260909 + automation ownership 260922). If a hook is added or removed this test should fail - // and be updated deliberately. - assert.equal(validateManifestTargets(root).length, 17); + // memory-write gate 260909 + automation ownership 260922 + the agent-thread + // PermissionRequest hook and SessionStart advisory 260928). If a hook is added or + // removed this test should fail and be updated deliberately. + assert.equal(validateManifestTargets(root).length, 19); }); // ---- D. doctor integration ------------------------------------------------- diff --git a/plugins/codexclaw/inventory.json b/plugins/codexclaw/inventory.json index d2c99abf..a468f6e2 100644 --- a/plugins/codexclaw/inventory.json +++ b/plugins/codexclaw/inventory.json @@ -135,6 +135,12 @@ "skills/qa/scripts/validate-evidence.mjs" ], "hooks": [ + { + "file": "permission-request-allowing-agent-thread.json", + "event": "PermissionRequest", + "component": "pabcd-state", + "matcher": "*" + }, { "file": "post-compact-injecting-bg-terminal-affordance.json", "event": "PostCompact", @@ -219,6 +225,12 @@ "component": "bg-wake", "matcher": null }, + { + "file": "session-start-advising-agent-thread-permissions.json", + "event": "SessionStart", + "component": "pabcd-state", + "matcher": null + }, { "file": "session-start-announcing-map-affordance.json", "event": "SessionStart", diff --git a/plugins/codexclaw/test/hook-e2e.test.mjs b/plugins/codexclaw/test/hook-e2e.test.mjs index de8f01de..adcd1643 100644 --- a/plugins/codexclaw/test/hook-e2e.test.mjs +++ b/plugins/codexclaw/test/hook-e2e.test.mjs @@ -227,7 +227,8 @@ test("WP7/G19: every manifest hook command resolves to an existing dist entrypoi // 260910: 25 -> 28 with bg-wake's three hooks. The pin is deliberate — it is the // machine-checked partner of the README badges and inventory.json, so an optional // component removes itself here too (see `cxc bg removal`). - assert.ok(Array.isArray(manifest.hooks) && manifest.hooks.length === 29, "expected 29 declared hooks"); + // 260928: 29 -> 31 with the agent-thread PermissionRequest hook and its SessionStart advisory. + assert.ok(Array.isArray(manifest.hooks) && manifest.hooks.length === 31, "expected 31 declared hooks"); for (const rel of manifest.hooks) { const { distAbs } = readHookCommand(rel); // Settle-retry: a concurrent rebuild (C10) may briefly unlink dist mid-run. From 7efa7ec396c199e8d04c5838376111ea796757a2 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 02:29:22 +0900 Subject: [PATCH 46/90] fix(pabcd-state): harden agent-thread permission config checks --- .../dist/agent-thread-permissions.js | 200 +++++++++++++++--- .../components/pabcd-state/dist/cli.js | 1 - .../src/agent-thread-permissions.ts | 200 +++++++++++++++--- .../components/pabcd-state/src/cli.ts | 1 - .../test/agent-thread-permissions.test.ts | 98 ++++++++- 5 files changed, 428 insertions(+), 72 deletions(-) diff --git a/plugins/codexclaw/components/pabcd-state/dist/agent-thread-permissions.js b/plugins/codexclaw/components/pabcd-state/dist/agent-thread-permissions.js index 9827c3e2..98f3e4f6 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/agent-thread-permissions.js +++ b/plugins/codexclaw/components/pabcd-state/dist/agent-thread-permissions.js @@ -53,27 +53,140 @@ function boundedText(path ) { return readFileSync(path, "utf8"); } -function globalOptIn(env , cwd ) { - const override = env.CODEXCLAW_HOME?.trim(); - if (override && !isAbsolute(override)) return false; - const home = override || join(homedir(), ".codexclaw"); - const configPath = join(home, "config.json"); +function trustedConfigText( + overrideValue , defaultDirectory , filename , cwd , +) { + const override = overrideValue?.trim(); + if (override && !isAbsolute(override)) return null; + const home = override || join(homedir(), defaultDirectory); + const configPath = join(home, filename); const realCwd = realpathSync(cwd); const realConfig = realpathSync(configPath); const rel = relative(realCwd, realConfig); const withinCwd = rel === "" || (!rel.startsWith("..") && !isAbsolute(rel)); const defaultAtHome = !override && realCwd === realpathSync(homedir()) && !lstatSync(configPath).isSymbolicLink() && lstatSync(configPath).isFile(); - if (withinCwd && !defaultAtHome) return false; - const config = object(JSON.parse(boundedText(realConfig) ?? "null")); + if (withinCwd && !defaultAtHome) return null; + return boundedText(realConfig); +} + +function globalOptIn(env , cwd ) { + const config = object(JSON.parse(trustedConfigText(env.CODEXCLAW_HOME, ".codexclaw", "config.json", cwd) ?? "null")); const permissions = object(config?.permissions); return permissions?.agentCreatedThreadAutoAllow === true; } + + +function table() { + return { kind: "table", children: new Map() }; +} + +function keyPath(source , start = 0) { + let index = start; + const parts = []; + const spaces = () => { while (source[index] === " " || source[index] === "\t") index += 1; }; + spaces(); + while (index < source.length) { + let part = ""; + const quote = source[index]; + if (quote === '"' || quote === "'") { + index += 1; + let closed = false; + while (index < source.length) { + const char = source[index++]; + if (char === quote) { closed = true; break; } + if (/[\x00-\x1f\x7f]/.test(char)) return null; + if (quote === '"' && char === "\\") { + const escape = source[index++]; + const simple = { b: "\b", t: "\t", n: "\n", f: "\f", r: "\r", '"': '"', "\\": "\\" }; + if (Object.hasOwn(simple, escape)) part += simple[escape]; + else if (escape === "u" || escape === "U") { + const length = escape === "u" ? 4 : 8; + const hex = source.slice(index, index + length); + if (!new RegExp(`^[0-9a-fA-F]{${length}}$`).test(hex)) return null; + const code = Number.parseInt(hex, 16); + if (code > 0x10ffff || (code >= 0xd800 && code <= 0xdfff)) return null; + part += String.fromCodePoint(code); + index += length; + } else return null; + } else part += char; + } + if (!closed) return null; + } else { + const match = /^[A-Za-z0-9_-]+/.exec(source.slice(index)); + if (!match) return null; + part = match[0]; + index += part.length; + } + parts.push(part); + spaces(); + if (source[index] !== ".") break; + index += 1; + spaces(); + if (index >= source.length) return null; + } + return parts.length ? { parts, end: index } : null; +} + +function assignKey(scope , parts ) { + let current = scope; + for (let index = 0; index < parts.length; index += 1) { + const children = current.children; + if (!children) return false; + const name = parts[index]; + const existing = children.get(name); + if (index === parts.length - 1) { + if (existing) return false; + children.set(name, { kind: "scalar" }); + return true; + } + if (!existing) { + const nested = table(); + children.set(name, nested); + current = nested; + } else { + if (existing.kind !== "table") return false; + current = existing; + } + } + return false; +} + +function enterTable(root , parts , array ) { + let current = root; + for (let index = 0; index < parts.length; index += 1) { + const children = current.children; + if (!children) return null; + const name = parts[index]; + let entry = children.get(name); + const last = index === parts.length - 1; + if (last && array) { + if (!entry) { entry = { kind: "array" }; children.set(name, entry); } + if (entry.kind !== "array") return null; + entry.latest = table(); + return entry.latest; + } + if (!entry) { entry = table(); children.set(name, entry); } + if (entry.kind === "array") { + if (!entry.latest) return null; + current = entry.latest; + } else if (entry.kind === "table") { + if (last) { + if (entry.declared) return null; + entry.declared = true; + } + current = entry; + } else return null; + } + return current; +} + /** A bounded structural scanner: unknown TOML is never permission evidence. */ function validValue(source ) { let index = 0; - const scalar = /^(?:true|false|[+-]?(?:0|[1-9](?:[0-9_]*[0-9])?)(?:\.[0-9_]+)?(?:[eE][+-]?[0-9_]+)?|[+-]?(?:inf|nan)|\d{4}-\d\d-\d\d(?:[Tt ]\d\d:\d\d:\d\d(?:\.\d+)?(?:[Zz]|[+-]\d\d:\d\d)?)?|\d\d:\d\d:\d\d(?:\.\d+)?|0[xX][0-9a-fA-F_]+|0[oO][0-7_]+|0[bB][01_]+)$/; + const digits = "[0-9](?:_?[0-9])*"; + const scalar = new RegExp(`^(?:true|false|[+-]?(?:0|[1-9](?:_?[0-9])*)(?:\\.${digits})?(?:[eE][+-]?${digits})?|[+-]?(?:inf|nan)|\\d{4}-\\d\\d-\\d\\d(?:[Tt ]\\d\\d:\\d\\d:\\d\\d(?:\\.\\d+)?(?:[Zz]|[+-]\\d\\d:\\d\\d)?)?|\\d\\d:\\d\\d:\\d\\d(?:\\.\\d+)?|0[xX][0-9a-fA-F](?:_?[0-9a-fA-F])*|0[oO][0-7](?:_?[0-7])*|0[bB][01](?:_?[01])*)$`); const skip = () => { while (index < source.length) { if (/\s/.test(source[index])) { index += 1; continue; } @@ -93,18 +206,30 @@ function validValue(source ) { while (index < source.length) { if (source.startsWith(mark, index)) { index += mark.length; return true; } if (!triple && /[\r\n]/.test(source[index])) return false; - if (quote === '"' && source[index] === "\\") index += 1; + if (/[\x00-\x08\x0b\x0e-\x1f\x7f]/.test(source[index])) return false; + if (quote === '"' && source[index] === "\\") { + index += 1; + const escape = source[index]; + if ("btnfr\"\\".includes(escape ?? "\0")) { index += 1; continue; } + if (escape === "u" || escape === "U") { + const length = escape === "u" ? 4 : 8; + const hex = source.slice(index + 1, index + 1 + length); + if (!new RegExp(`^[0-9a-fA-F]{${length}}$`).test(hex)) return false; + const code = Number.parseInt(hex, 16); + if (code > 0x10ffff || (code >= 0xd800 && code <= 0xdfff)) return false; + index += length + 1; + continue; + } + if (triple) { + const continuation = /^[ \t]*(?:\r?\n)/.exec(source.slice(index)); + if (continuation) { index += continuation[0].length; continue; } + } + return false; + } index += 1; } return false; }; - const key = () => { - if (source[index] === '"' || source[index] === "'") return quoted(); - const match = /^[A-Za-z0-9_-]+/.exec(source.slice(index)); - if (!match) return false; - index += match[0].length; - return true; - }; const value = (depth ) => { if (depth > 64) return false; skip(); @@ -113,11 +238,14 @@ function validValue(source ) { if (opener === "[" || opener === "{") { index += 1; const closer = opener === "[" ? "]" : "}"; + const inline = opener === "{" ? table() : null; skip(); if (source[index] === closer) { index += 1; return true; } while (index < source.length) { if (opener === "{") { - if (!key()) return false; + const parsed = keyPath(source, index); + if (!parsed || !assignKey(inline , parsed.parts)) return false; + index = parsed.end; skip(); if (source[index++] !== "=") return false; } @@ -142,28 +270,34 @@ function validValue(source ) { function validTomlAndTopLevel(content ) { const seen = new Map (); - let inTopLevel = true; + const root = table(); + let current = root; const lines = content.replace(/^\uFEFF/, "").split(/\r?\n/); for (let index = 0; index < lines.length; index += 1) { const line = lines[index].trim(); if (line === "" || line.startsWith("#")) continue; if (line.startsWith("[")) { - if (!/^(?:\[\[[^[\]\r\n]+\]\]|\[[^[\]\r\n]+\])\s*(?:#.*)?$/.test(line)) return false; - inTopLevel = false; + const array = line.startsWith("[["); + const header = array ? /^\[\[(.*)\]\]\s*(?:#.*)?$/.exec(line) : /^\[(.*)\]\s*(?:#.*)?$/.exec(line); + if (!header) return false; + const parsed = keyPath(header[1]); + if (!parsed || parsed.end !== header[1].length) return false; + const next = enterTable(root, parsed.parts, array); + if (!next) return false; + current = next; continue; } - const assignment = /^([A-Za-z0-9_-]+|"[^"\r\n]+"|'[^'\r\n]+')\s*=\s*(.*)$/.exec(line); - if (!assignment) return false; - const key = assignment[1].replace(/^["']|["']$/g, ""); - if (inTopLevel && key === "profile") return false; - let value = assignment[2]; + const parsed = keyPath(line); + if (!parsed || line[parsed.end] !== "=" || !assignKey(current, parsed.parts)) return false; + const key = parsed.parts.length === 1 ? parsed.parts[0] : ""; + if (current === root && key === "profile") return false; + let value = line.slice(parsed.end + 1).trimStart(); while (!validValue(value)) { // Only arrays, inline tables and triple strings may continue onto another line. if (!/^(?:\[|\{|"""|''')/.test(value) || index + 1 >= lines.length) return false; value += `\n${lines[++index]}`; } - if (inTopLevel && (key === "approval_policy" || key === "sandbox_mode")) { - if (seen.has(key)) return false; + if (current === root && (key === "approval_policy" || key === "sandbox_mode")) { const exact = /^"([^"\r\n]*)"\s*(?:#.*)?$/.exec(value); if (!exact) return false; seen.set(key, exact[1]); @@ -173,11 +307,8 @@ function validTomlAndTopLevel(content ) { seen.get("sandbox_mode") === "danger-full-access"; } -function codexConfigFullAccess(env ) { - const override = env.CODEX_HOME?.trim(); - if (override && !isAbsolute(override)) return false; - const home = override || join(homedir(), ".codex"); - const content = boundedText(join(home, "config.toml")); +function codexConfigFullAccess(env , cwd ) { + const content = trustedConfigText(env.CODEX_HOME, ".codex", "config.toml", cwd); if (content === null) return false; return validTomlAndTopLevel(content); } @@ -206,7 +337,7 @@ export function handleAgentThreadPermissionRequest( const input = parseHook(raw, "PermissionRequest"); return input && typeof input.cwd === "string" && input.cwd !== "" && coveredTool(input.tool_name) && globalOptIn(env, input.cwd) && - agentCreatedRoot(input) && codexConfigFullAccess(env) ? ALLOW : ""; + agentCreatedRoot(input) && codexConfigFullAccess(env, input.cwd) ? ALLOW : ""; } catch { return ""; } @@ -217,7 +348,8 @@ export function handleAgentThreadSessionStartAdvisory( ) { try { const input = parseHook(raw, "SessionStart"); - if (!input || !agentCreatedRoot(input) || !codexConfigFullAccess(env)) return ""; + if (!input || typeof input.cwd !== "string" || input.cwd === "" || + !agentCreatedRoot(input) || !codexConfigFullAccess(env, input.cwd)) return ""; return `${JSON.stringify({ systemMessage: USER_ADVICE, hookSpecificOutput: { hookEventName: "SessionStart", additionalContext: MODEL_ADVICE }, diff --git a/plugins/codexclaw/components/pabcd-state/dist/cli.js b/plugins/codexclaw/components/pabcd-state/dist/cli.js index 4df4207f..a0eaedd6 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/cli.js +++ b/plugins/codexclaw/components/pabcd-state/dist/cli.js @@ -350,7 +350,6 @@ async function main() { const raw = stdin.raw; if (event === "permission-request" || event === "session-start-permission-advisory") { try { - recordHookInvocation(raw, "pabcd-state", event, import.meta.url); const { handleAgentThreadPermissionRequest, handleAgentThreadSessionStartAdvisory } = await import("./agent-thread-permissions.js"); const result = event === "permission-request" diff --git a/plugins/codexclaw/components/pabcd-state/src/agent-thread-permissions.ts b/plugins/codexclaw/components/pabcd-state/src/agent-thread-permissions.ts index c7a7ebe2..9c435cde 100644 --- a/plugins/codexclaw/components/pabcd-state/src/agent-thread-permissions.ts +++ b/plugins/codexclaw/components/pabcd-state/src/agent-thread-permissions.ts @@ -53,27 +53,140 @@ function boundedText(path: string): string | null { return readFileSync(path, "utf8"); } -function globalOptIn(env: NodeJS.ProcessEnv, cwd: string): boolean { - const override = env.CODEXCLAW_HOME?.trim(); - if (override && !isAbsolute(override)) return false; - const home = override || join(homedir(), ".codexclaw"); - const configPath = join(home, "config.json"); +function trustedConfigText( + overrideValue: string | undefined, defaultDirectory: string, filename: string, cwd: string, +): string | null { + const override = overrideValue?.trim(); + if (override && !isAbsolute(override)) return null; + const home = override || join(homedir(), defaultDirectory); + const configPath = join(home, filename); const realCwd = realpathSync(cwd); const realConfig = realpathSync(configPath); const rel = relative(realCwd, realConfig); const withinCwd = rel === "" || (!rel.startsWith("..") && !isAbsolute(rel)); const defaultAtHome = !override && realCwd === realpathSync(homedir()) && !lstatSync(configPath).isSymbolicLink() && lstatSync(configPath).isFile(); - if (withinCwd && !defaultAtHome) return false; - const config = object(JSON.parse(boundedText(realConfig) ?? "null")); + if (withinCwd && !defaultAtHome) return null; + return boundedText(realConfig); +} + +function globalOptIn(env: NodeJS.ProcessEnv, cwd: string): boolean { + const config = object(JSON.parse(trustedConfigText(env.CODEXCLAW_HOME, ".codexclaw", "config.json", cwd) ?? "null")); const permissions = object(config?.permissions); return permissions?.agentCreatedThreadAutoAllow === true; } +type TomlEntry = { kind: "table" | "array" | "scalar"; children?: Map; declared?: boolean; latest?: TomlEntry }; + +function table(): TomlEntry { + return { kind: "table", children: new Map() }; +} + +function keyPath(source: string, start = 0): { parts: string[]; end: number } | null { + let index = start; + const parts: string[] = []; + const spaces = (): void => { while (source[index] === " " || source[index] === "\t") index += 1; }; + spaces(); + while (index < source.length) { + let part = ""; + const quote = source[index]; + if (quote === '"' || quote === "'") { + index += 1; + let closed = false; + while (index < source.length) { + const char = source[index++]; + if (char === quote) { closed = true; break; } + if (/[\x00-\x1f\x7f]/.test(char)) return null; + if (quote === '"' && char === "\\") { + const escape = source[index++]; + const simple: Record = { b: "\b", t: "\t", n: "\n", f: "\f", r: "\r", '"': '"', "\\": "\\" }; + if (Object.hasOwn(simple, escape)) part += simple[escape]; + else if (escape === "u" || escape === "U") { + const length = escape === "u" ? 4 : 8; + const hex = source.slice(index, index + length); + if (!new RegExp(`^[0-9a-fA-F]{${length}}$`).test(hex)) return null; + const code = Number.parseInt(hex, 16); + if (code > 0x10ffff || (code >= 0xd800 && code <= 0xdfff)) return null; + part += String.fromCodePoint(code); + index += length; + } else return null; + } else part += char; + } + if (!closed) return null; + } else { + const match = /^[A-Za-z0-9_-]+/.exec(source.slice(index)); + if (!match) return null; + part = match[0]; + index += part.length; + } + parts.push(part); + spaces(); + if (source[index] !== ".") break; + index += 1; + spaces(); + if (index >= source.length) return null; + } + return parts.length ? { parts, end: index } : null; +} + +function assignKey(scope: TomlEntry, parts: string[]): boolean { + let current = scope; + for (let index = 0; index < parts.length; index += 1) { + const children = current.children; + if (!children) return false; + const name = parts[index]; + const existing = children.get(name); + if (index === parts.length - 1) { + if (existing) return false; + children.set(name, { kind: "scalar" }); + return true; + } + if (!existing) { + const nested = table(); + children.set(name, nested); + current = nested; + } else { + if (existing.kind !== "table") return false; + current = existing; + } + } + return false; +} + +function enterTable(root: TomlEntry, parts: string[], array: boolean): TomlEntry | null { + let current = root; + for (let index = 0; index < parts.length; index += 1) { + const children = current.children; + if (!children) return null; + const name = parts[index]; + let entry = children.get(name); + const last = index === parts.length - 1; + if (last && array) { + if (!entry) { entry = { kind: "array" }; children.set(name, entry); } + if (entry.kind !== "array") return null; + entry.latest = table(); + return entry.latest; + } + if (!entry) { entry = table(); children.set(name, entry); } + if (entry.kind === "array") { + if (!entry.latest) return null; + current = entry.latest; + } else if (entry.kind === "table") { + if (last) { + if (entry.declared) return null; + entry.declared = true; + } + current = entry; + } else return null; + } + return current; +} + /** A bounded structural scanner: unknown TOML is never permission evidence. */ function validValue(source: string): boolean { let index = 0; - const scalar = /^(?:true|false|[+-]?(?:0|[1-9](?:[0-9_]*[0-9])?)(?:\.[0-9_]+)?(?:[eE][+-]?[0-9_]+)?|[+-]?(?:inf|nan)|\d{4}-\d\d-\d\d(?:[Tt ]\d\d:\d\d:\d\d(?:\.\d+)?(?:[Zz]|[+-]\d\d:\d\d)?)?|\d\d:\d\d:\d\d(?:\.\d+)?|0[xX][0-9a-fA-F_]+|0[oO][0-7_]+|0[bB][01_]+)$/; + const digits = "[0-9](?:_?[0-9])*"; + const scalar = new RegExp(`^(?:true|false|[+-]?(?:0|[1-9](?:_?[0-9])*)(?:\\.${digits})?(?:[eE][+-]?${digits})?|[+-]?(?:inf|nan)|\\d{4}-\\d\\d-\\d\\d(?:[Tt ]\\d\\d:\\d\\d:\\d\\d(?:\\.\\d+)?(?:[Zz]|[+-]\\d\\d:\\d\\d)?)?|\\d\\d:\\d\\d:\\d\\d(?:\\.\\d+)?|0[xX][0-9a-fA-F](?:_?[0-9a-fA-F])*|0[oO][0-7](?:_?[0-7])*|0[bB][01](?:_?[01])*)$`); const skip = (): void => { while (index < source.length) { if (/\s/.test(source[index])) { index += 1; continue; } @@ -93,18 +206,30 @@ function validValue(source: string): boolean { while (index < source.length) { if (source.startsWith(mark, index)) { index += mark.length; return true; } if (!triple && /[\r\n]/.test(source[index])) return false; - if (quote === '"' && source[index] === "\\") index += 1; + if (/[\x00-\x08\x0b\x0e-\x1f\x7f]/.test(source[index])) return false; + if (quote === '"' && source[index] === "\\") { + index += 1; + const escape = source[index]; + if ("btnfr\"\\".includes(escape ?? "\0")) { index += 1; continue; } + if (escape === "u" || escape === "U") { + const length = escape === "u" ? 4 : 8; + const hex = source.slice(index + 1, index + 1 + length); + if (!new RegExp(`^[0-9a-fA-F]{${length}}$`).test(hex)) return false; + const code = Number.parseInt(hex, 16); + if (code > 0x10ffff || (code >= 0xd800 && code <= 0xdfff)) return false; + index += length + 1; + continue; + } + if (triple) { + const continuation = /^[ \t]*(?:\r?\n)/.exec(source.slice(index)); + if (continuation) { index += continuation[0].length; continue; } + } + return false; + } index += 1; } return false; }; - const key = (): boolean => { - if (source[index] === '"' || source[index] === "'") return quoted(); - const match = /^[A-Za-z0-9_-]+/.exec(source.slice(index)); - if (!match) return false; - index += match[0].length; - return true; - }; const value = (depth: number): boolean => { if (depth > 64) return false; skip(); @@ -113,11 +238,14 @@ function validValue(source: string): boolean { if (opener === "[" || opener === "{") { index += 1; const closer = opener === "[" ? "]" : "}"; + const inline = opener === "{" ? table() : null; skip(); if (source[index] === closer) { index += 1; return true; } while (index < source.length) { if (opener === "{") { - if (!key()) return false; + const parsed = keyPath(source, index); + if (!parsed || !assignKey(inline!, parsed.parts)) return false; + index = parsed.end; skip(); if (source[index++] !== "=") return false; } @@ -142,28 +270,34 @@ function validValue(source: string): boolean { function validTomlAndTopLevel(content: string): boolean { const seen = new Map(); - let inTopLevel = true; + const root = table(); + let current = root; const lines = content.replace(/^\uFEFF/, "").split(/\r?\n/); for (let index = 0; index < lines.length; index += 1) { const line = lines[index].trim(); if (line === "" || line.startsWith("#")) continue; if (line.startsWith("[")) { - if (!/^(?:\[\[[^[\]\r\n]+\]\]|\[[^[\]\r\n]+\])\s*(?:#.*)?$/.test(line)) return false; - inTopLevel = false; + const array = line.startsWith("[["); + const header = array ? /^\[\[(.*)\]\]\s*(?:#.*)?$/.exec(line) : /^\[(.*)\]\s*(?:#.*)?$/.exec(line); + if (!header) return false; + const parsed = keyPath(header[1]); + if (!parsed || parsed.end !== header[1].length) return false; + const next = enterTable(root, parsed.parts, array); + if (!next) return false; + current = next; continue; } - const assignment = /^([A-Za-z0-9_-]+|"[^"\r\n]+"|'[^'\r\n]+')\s*=\s*(.*)$/.exec(line); - if (!assignment) return false; - const key = assignment[1].replace(/^["']|["']$/g, ""); - if (inTopLevel && key === "profile") return false; - let value = assignment[2]; + const parsed = keyPath(line); + if (!parsed || line[parsed.end] !== "=" || !assignKey(current, parsed.parts)) return false; + const key = parsed.parts.length === 1 ? parsed.parts[0] : ""; + if (current === root && key === "profile") return false; + let value = line.slice(parsed.end + 1).trimStart(); while (!validValue(value)) { // Only arrays, inline tables and triple strings may continue onto another line. if (!/^(?:\[|\{|"""|''')/.test(value) || index + 1 >= lines.length) return false; value += `\n${lines[++index]}`; } - if (inTopLevel && (key === "approval_policy" || key === "sandbox_mode")) { - if (seen.has(key)) return false; + if (current === root && (key === "approval_policy" || key === "sandbox_mode")) { const exact = /^"([^"\r\n]*)"\s*(?:#.*)?$/.exec(value); if (!exact) return false; seen.set(key, exact[1]); @@ -173,11 +307,8 @@ function validTomlAndTopLevel(content: string): boolean { seen.get("sandbox_mode") === "danger-full-access"; } -function codexConfigFullAccess(env: NodeJS.ProcessEnv): boolean { - const override = env.CODEX_HOME?.trim(); - if (override && !isAbsolute(override)) return false; - const home = override || join(homedir(), ".codex"); - const content = boundedText(join(home, "config.toml")); +function codexConfigFullAccess(env: NodeJS.ProcessEnv, cwd: string): boolean { + const content = trustedConfigText(env.CODEX_HOME, ".codex", "config.toml", cwd); if (content === null) return false; return validTomlAndTopLevel(content); } @@ -206,7 +337,7 @@ export function handleAgentThreadPermissionRequest( const input = parseHook(raw, "PermissionRequest"); return input && typeof input.cwd === "string" && input.cwd !== "" && coveredTool(input.tool_name) && globalOptIn(env, input.cwd) && - agentCreatedRoot(input) && codexConfigFullAccess(env) ? ALLOW : ""; + agentCreatedRoot(input) && codexConfigFullAccess(env, input.cwd) ? ALLOW : ""; } catch { return ""; } @@ -217,7 +348,8 @@ export function handleAgentThreadSessionStartAdvisory( ): string { try { const input = parseHook(raw, "SessionStart"); - if (!input || !agentCreatedRoot(input) || !codexConfigFullAccess(env)) return ""; + if (!input || typeof input.cwd !== "string" || input.cwd === "" || + !agentCreatedRoot(input) || !codexConfigFullAccess(env, input.cwd)) return ""; return `${JSON.stringify({ systemMessage: USER_ADVICE, hookSpecificOutput: { hookEventName: "SessionStart", additionalContext: MODEL_ADVICE }, diff --git a/plugins/codexclaw/components/pabcd-state/src/cli.ts b/plugins/codexclaw/components/pabcd-state/src/cli.ts index d88102ff..88740b1c 100644 --- a/plugins/codexclaw/components/pabcd-state/src/cli.ts +++ b/plugins/codexclaw/components/pabcd-state/src/cli.ts @@ -350,7 +350,6 @@ async function main(): Promise { const raw = stdin.raw; if (event === "permission-request" || event === "session-start-permission-advisory") { try { - recordHookInvocation(raw, "pabcd-state", event, import.meta.url); const { handleAgentThreadPermissionRequest, handleAgentThreadSessionStartAdvisory } = await import("./agent-thread-permissions.ts"); const result = event === "permission-request" diff --git a/plugins/codexclaw/components/pabcd-state/test/agent-thread-permissions.test.ts b/plugins/codexclaw/components/pabcd-state/test/agent-thread-permissions.test.ts index 36d66660..1b64ea20 100644 --- a/plugins/codexclaw/components/pabcd-state/test/agent-thread-permissions.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/agent-thread-permissions.test.ts @@ -1,7 +1,7 @@ import { test } from "node:test"; import assert from "node:assert/strict"; import { spawnSync } from "node:child_process"; -import { existsSync, mkdtempSync, mkdirSync, rmSync, symlinkSync, writeFileSync } from "node:fs"; +import { existsSync, mkdtempSync, mkdirSync, readdirSync, rmSync, symlinkSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; import { dirname, join, resolve } from "node:path"; import { fileURLToPath } from "node:url"; @@ -121,6 +121,90 @@ test("rejects unknown or conflicting Codex config", (t) => { assert.equal(f.send(), "", "oversized Codex config"); }); +for (const [name, suffix] of [ + ["invalid header key", "[bad key]\nx = true\n"], + ["invalid numeric underscores", "foo = 1__2\n"], + ["invalid basic string escape", 'foo = "\\q"\n'], + ["duplicate unrelated key", "foo = 1\nfoo = 2\n"], +] as const) { + test(`rejects ${name} anywhere in Codex config`, (t) => { + const f = fixture(t); + const variants = name === "invalid header key" ? [suffix, "[x. bad key]\ny = 1\n", "[[bad key]]\ny = 1\n"] + : name === "invalid numeric underscores" ? [suffix, ...["1_", "1._2", "1.2__3", "1e_2", "1e2_", "0x_1", "0x1__2", "0o7_", "0b1__0"].map((value) => `foo = ${value}\n`)] + : name === "invalid basic string escape" ? [suffix, 'foo = """\\q"""\n', '"\\q" = 1\n'] + : [suffix, 'foo = 1\n"foo" = 2\n', '[x]\na = 1\n"a" = 2\n', 'a = 1\na.b = 2\n']; + for (const variant of variants) { + writeFileSync(join(f.codexHome, "config.toml"), FULL + variant); + assert.equal(f.send(), "", variant); + assert.equal(f.advise(), "", variant); + } + }); +} + +test("accepts real dotted tables, literal keys and multiline roots while rejecting redefined scalar", (t) => { + const f = fixture(t); + const path = join(f.codexHome, "config.toml"); + writeFileSync(path, `${FULL}[features]\nsearch = true\nalpha . 'beta gamma' = 2\n[projects."/a/b"]\ntrust_level = "trusted"\n[hooks.state."x@y:z.json:pre_tool_use:0:0"]\nenabled = true\n[sandbox_workspace_write]\nwritable_roots = [\n "/tmp",\n "/var/tmp",\n]\n[literal.'raw key']\nvalue = 1\n`); + assert.equal(f.send(), ALLOW); + assert.notEqual(f.advise(), ""); + writeFileSync(path, `${FULL}[numbers]\nint = 1_000\nfloat = 1_000.2_50e+1_0\nhex = 0xA_B\noctal = 0o7_1\nbinary = 0b1_0\nmultiline = """hello\\\n world"""\n`); + assert.equal(f.send(), ALLOW); + writeFileSync(path, `${FULL}foo = 1\nfoo.bar = 2\n`); + assert.equal(f.send(), ""); +}); + +test("allows repeated array tables with fresh keys but rejects repeated standard tables", (t) => { + const f = fixture(t); + const path = join(f.codexHome, "config.toml"); + writeFileSync(path, `${FULL}[[x]]\nkey = 1\n[[x]]\nkey = 2\n`); + assert.equal(f.send(), ALLOW); + writeFileSync(path, `${FULL}[x]\nkey = 1\n[x]\nother = 2\n`); + assert.equal(f.send(), ""); +}); + +test("project-controlled CODEX_HOME config cannot grant permission or advisory", (t) => { + const f = fixture(t); + const home = join(f.cwd, "codex"); + mkdirSync(home); + writeFileSync(join(home, "config.toml"), FULL); + const env = { ...f.env, CODEX_HOME: home }; + assert.equal(f.send({}, env), ""); + assert.equal(f.advise({}, env), ""); +}); + +test("symlinked Codex config into project cannot grant permission or advisory", (t) => { + const f = fixture(t); + const path = join(f.cwd, "config.toml"); + writeFileSync(path, FULL); + rmSync(join(f.codexHome, "config.toml")); + symlinkSync(path, join(f.codexHome, "config.toml")); + assert.equal(f.send(), ""); + assert.equal(f.advise(), ""); +}); + +test("advisory requires string cwd even with outside Codex config", (t) => { + const f = fixture(t); + assert.notEqual(f.advise(), ""); + assert.equal(f.advise({ cwd: null }), ""); +}); + +test("permission CLI verbs leave project CODEX_HOME untouched", (t) => { + const f = fixture(t); + const home = join(f.cwd, "codex"); + mkdirSync(home); + writeFileSync(join(home, "config.toml"), FULL); + const env = { ...f.env, CODEX_HOME: home }; + const before = readdirSync(f.cwd).sort(); + const homeBefore = readdirSync(home).sort(); + for (const [verb, hook_event_name] of [["permission-request", "PermissionRequest"], ["session-start-permission-advisory", "SessionStart"]] as const) { + const result = cli(verb, { ...f.input, hook_event_name }, f.cwd, env); + assert.equal(result.status, 0, result.stderr); + assert.equal(result.stdout, ""); + } + assert.deepEqual(readdirSync(f.cwd).sort(), before); + assert.deepEqual(readdirSync(home).sort(), homeBefore); +}); + test("ignores project-local opt-in and noncovered tool", (t) => { const f = fixture(t); writeFileSync(join(f.clawHome, "config.json"), "{}"); @@ -161,10 +245,20 @@ test("default home config remains eligible at home but a project symlink does no mkdirSync(defaultDir); const defaultConfig = join(defaultDir, "config.json"); writeFileSync(defaultConfig, '{"permissions":{"agentCreatedThreadAutoAllow":true}}'); - const env = { CODEX_HOME: f.codexHome, CODEXCLAW_HOME: "", HOME: f.root }; + const codexDir = join(f.root, ".codex"); + mkdirSync(codexDir); + writeFileSync(join(codexDir, "config.toml"), FULL); + const env = { CODEX_HOME: "", CODEXCLAW_HOME: "", HOME: f.root }; const atHome = cli("permission-request", { ...f.input, cwd: f.root }, f.root, env); assert.equal(atHome.status, 0, atHome.stderr); assert.equal(atHome.stdout, ALLOW); + const projectToml = join(f.cwd, "config.toml"); + writeFileSync(projectToml, FULL); + rmSync(join(codexDir, "config.toml")); + symlinkSync(projectToml, join(codexDir, "config.toml")); + assert.equal(cli("permission-request", { ...f.input, cwd: f.root }, f.root, env).stdout, ""); + rmSync(join(codexDir, "config.toml")); + writeFileSync(join(codexDir, "config.toml"), FULL); const projectConfig = join(f.cwd, "config.json"); writeFileSync(projectConfig, '{"permissions":{"agentCreatedThreadAutoAllow":true}}'); rmSync(defaultConfig); From 41a69f24420fca273375407d08eb557ca26014b4 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 02:29:43 +0900 Subject: [PATCH 47/90] docs: publish measured test count (3677) --- README.ko.md | 2 +- README.md | 2 +- README.zh.md | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/README.ko.md b/README.ko.md index 9fbd0870..6e1e10d4 100644 --- a/README.ko.md +++ b/README.ko.md @@ -13,7 +13,7 @@

CI - 3,667 tests + 3,677 tests 29 skills 31 hooks Documentation diff --git a/README.md b/README.md index 22720140..84031047 100644 --- a/README.md +++ b/README.md @@ -13,7 +13,7 @@

CI - 3,667 tests + 3,677 tests 29 skills 31 hooks Documentation diff --git a/README.zh.md b/README.zh.md index a2020ae8..d25ae253 100644 --- a/README.zh.md +++ b/README.zh.md @@ -13,7 +13,7 @@

CI - 3,667 tests + 3,677 tests 29 skills 31 hooks Documentation From ff7b6809351c110cf40f61f0ff138511916c5944 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 02:34:11 +0900 Subject: [PATCH 48/90] fix(pabcd-state): validate TOML dates and inline tables; home exception only for literal default paths --- .../dist/agent-thread-permissions.js | 38 +++++++++++++++-- .../src/agent-thread-permissions.ts | 38 +++++++++++++++-- .../test/agent-thread-permissions.test.ts | 42 +++++++++++++++++++ 3 files changed, 110 insertions(+), 8 deletions(-) diff --git a/plugins/codexclaw/components/pabcd-state/dist/agent-thread-permissions.js b/plugins/codexclaw/components/pabcd-state/dist/agent-thread-permissions.js index 98f3e4f6..03fda01b 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/agent-thread-permissions.js +++ b/plugins/codexclaw/components/pabcd-state/dist/agent-thread-permissions.js @@ -53,6 +53,24 @@ function boundedText(path ) { return readFileSync(path, "utf8"); } +/** Dates and times must name a real calendar day and clock time. */ +function validDateTime(token ) { + const date = /^(\d{4})-(\d\d)-(\d\d)/.exec(token); + if (date) { + const year = Number(date[1]); + const month = Number(date[2]); + const day = Number(date[3]); + if (month < 1 || month > 12) return false; + if (day < 1 || day > new Date(Date.UTC(year, month, 0)).getUTCDate()) return false; + } + const time = /(?:^|[Tt ])(\d\d):(\d\d):(\d\d)(?:\.\d+)?(?:[Zz]|[+-](\d\d):(\d\d))?$/.exec(token); + if (time) { + if (Number(time[1]) > 23 || Number(time[2]) > 59 || Number(time[3]) > 60) return false; + if (time[4] !== undefined && (Number(time[4]) > 23 || Number(time[5]) > 59)) return false; + } + return true; +} + function trustedConfigText( overrideValue , defaultDirectory , filename , cwd , ) { @@ -64,7 +82,12 @@ function trustedConfigText( const realConfig = realpathSync(configPath); const rel = relative(realCwd, realConfig); const withinCwd = rel === "" || (!rel.startsWith("..") && !isAbsolute(rel)); - const defaultAtHome = !override && realCwd === realpathSync(homedir()) && + // The home-cwd exception covers only the literal default file: if the default + // directory or file is a symlink anywhere, its real path differs and it is judged + // like any other file inside cwd. + const realHome = realpathSync(homedir()); + const defaultAtHome = !override && realCwd === realHome && + realConfig === join(realHome, defaultDirectory, filename) && !lstatSync(configPath).isSymbolicLink() && lstatSync(configPath).isFile(); if (withinCwd && !defaultAtHome) return null; return boundedText(realConfig); @@ -230,17 +253,24 @@ function validValue(source ) { } return false; }; + // TOML 1.0 inline tables stay on one line (newlines only inside string values). + const singleLineInline = (start ) => { + const body = source.slice(start, index) + .replace(/"""[\s\S]*?"""|'''[\s\S]*?'''|"(?:[^"\\\n]|\\.)*"|'[^'\n]*'/g, ""); + return !/[\r\n]/.test(body); + }; const value = (depth ) => { if (depth > 64) return false; skip(); const opener = source[index]; if (opener === '"' || opener === "'") return quoted(); if (opener === "[" || opener === "{") { + const start = index; index += 1; const closer = opener === "[" ? "]" : "}"; const inline = opener === "{" ? table() : null; skip(); - if (source[index] === closer) { index += 1; return true; } + if (source[index] === closer) { index += 1; return opener === "[" || singleLineInline(start); } while (index < source.length) { if (opener === "{") { const parsed = keyPath(source, index); @@ -251,7 +281,7 @@ function validValue(source ) { } if (!value(depth + 1)) return false; skip(); - if (source[index] === closer) { index += 1; return true; } + if (source[index] === closer) { index += 1; return opener === "[" || singleLineInline(start); } if (source[index++] !== ",") return false; skip(); if (opener === "[" && source[index] === closer) { index += 1; return true; } @@ -259,7 +289,7 @@ function validValue(source ) { return false; } const bare = /^[^\s,\]\}#]+/.exec(source.slice(index)); - if (!bare || !scalar.test(bare[0])) return false; + if (!bare || !scalar.test(bare[0]) || !validDateTime(bare[0])) return false; index += bare[0].length; return true; }; diff --git a/plugins/codexclaw/components/pabcd-state/src/agent-thread-permissions.ts b/plugins/codexclaw/components/pabcd-state/src/agent-thread-permissions.ts index 9c435cde..73611252 100644 --- a/plugins/codexclaw/components/pabcd-state/src/agent-thread-permissions.ts +++ b/plugins/codexclaw/components/pabcd-state/src/agent-thread-permissions.ts @@ -53,6 +53,24 @@ function boundedText(path: string): string | null { return readFileSync(path, "utf8"); } +/** Dates and times must name a real calendar day and clock time. */ +function validDateTime(token: string): boolean { + const date = /^(\d{4})-(\d\d)-(\d\d)/.exec(token); + if (date) { + const year = Number(date[1]); + const month = Number(date[2]); + const day = Number(date[3]); + if (month < 1 || month > 12) return false; + if (day < 1 || day > new Date(Date.UTC(year, month, 0)).getUTCDate()) return false; + } + const time = /(?:^|[Tt ])(\d\d):(\d\d):(\d\d)(?:\.\d+)?(?:[Zz]|[+-](\d\d):(\d\d))?$/.exec(token); + if (time) { + if (Number(time[1]) > 23 || Number(time[2]) > 59 || Number(time[3]) > 60) return false; + if (time[4] !== undefined && (Number(time[4]) > 23 || Number(time[5]) > 59)) return false; + } + return true; +} + function trustedConfigText( overrideValue: string | undefined, defaultDirectory: string, filename: string, cwd: string, ): string | null { @@ -64,7 +82,12 @@ function trustedConfigText( const realConfig = realpathSync(configPath); const rel = relative(realCwd, realConfig); const withinCwd = rel === "" || (!rel.startsWith("..") && !isAbsolute(rel)); - const defaultAtHome = !override && realCwd === realpathSync(homedir()) && + // The home-cwd exception covers only the literal default file: if the default + // directory or file is a symlink anywhere, its real path differs and it is judged + // like any other file inside cwd. + const realHome = realpathSync(homedir()); + const defaultAtHome = !override && realCwd === realHome && + realConfig === join(realHome, defaultDirectory, filename) && !lstatSync(configPath).isSymbolicLink() && lstatSync(configPath).isFile(); if (withinCwd && !defaultAtHome) return null; return boundedText(realConfig); @@ -230,17 +253,24 @@ function validValue(source: string): boolean { } return false; }; + // TOML 1.0 inline tables stay on one line (newlines only inside string values). + const singleLineInline = (start: number): boolean => { + const body = source.slice(start, index) + .replace(/"""[\s\S]*?"""|'''[\s\S]*?'''|"(?:[^"\\\n]|\\.)*"|'[^'\n]*'/g, ""); + return !/[\r\n]/.test(body); + }; const value = (depth: number): boolean => { if (depth > 64) return false; skip(); const opener = source[index]; if (opener === '"' || opener === "'") return quoted(); if (opener === "[" || opener === "{") { + const start = index; index += 1; const closer = opener === "[" ? "]" : "}"; const inline = opener === "{" ? table() : null; skip(); - if (source[index] === closer) { index += 1; return true; } + if (source[index] === closer) { index += 1; return opener === "[" || singleLineInline(start); } while (index < source.length) { if (opener === "{") { const parsed = keyPath(source, index); @@ -251,7 +281,7 @@ function validValue(source: string): boolean { } if (!value(depth + 1)) return false; skip(); - if (source[index] === closer) { index += 1; return true; } + if (source[index] === closer) { index += 1; return opener === "[" || singleLineInline(start); } if (source[index++] !== ",") return false; skip(); if (opener === "[" && source[index] === closer) { index += 1; return true; } @@ -259,7 +289,7 @@ function validValue(source: string): boolean { return false; } const bare = /^[^\s,\]\}#]+/.exec(source.slice(index)); - if (!bare || !scalar.test(bare[0])) return false; + if (!bare || !scalar.test(bare[0]) || !validDateTime(bare[0])) return false; index += bare[0].length; return true; }; diff --git a/plugins/codexclaw/components/pabcd-state/test/agent-thread-permissions.test.ts b/plugins/codexclaw/components/pabcd-state/test/agent-thread-permissions.test.ts index 1b64ea20..6f7a0d8b 100644 --- a/plugins/codexclaw/components/pabcd-state/test/agent-thread-permissions.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/agent-thread-permissions.test.ts @@ -44,6 +44,48 @@ test("allows opted-in agent-created root thread for covered tools", (t) => { assert.equal(f.send({ tool_name: "Bash", tool_input: { description: "network-access example.com" } }), ALLOW); }); + +test("symlinked default directories into the home-cwd project cannot grant permission", (t) => { + const f = fixture(t); + const env = { CODEX_HOME: "", CODEXCLAW_HOME: "", HOME: f.root }; + // Real project directories under the home, reached through symlinked defaults. + const projectCodex = join(f.cwd, ".codex"); + const projectClaw = join(f.cwd, ".codexclaw"); + mkdirSync(projectCodex); + mkdirSync(projectClaw); + writeFileSync(join(projectCodex, "config.toml"), FULL); + writeFileSync(join(projectClaw, "config.json"), '{"permissions":{"agentCreatedThreadAutoAllow":true}}'); + const codexDir = join(f.root, ".codex"); + const clawDir = join(f.root, ".codexclaw"); + mkdirSync(clawDir); + writeFileSync(join(clawDir, "config.json"), '{"permissions":{"agentCreatedThreadAutoAllow":true}}'); + symlinkSync(projectCodex, codexDir); + assert.equal(cli("permission-request", { ...f.input, cwd: f.root }, f.root, env).stdout, "", "symlinked ~/.codex"); + rmSync(codexDir); + mkdirSync(codexDir); + writeFileSync(join(codexDir, "config.toml"), FULL); + rmSync(clawDir, { recursive: true, force: true }); + symlinkSync(projectClaw, clawDir); + assert.equal(cli("permission-request", { ...f.input, cwd: f.root }, f.root, env).stdout, "", "symlinked ~/.codexclaw"); +}); + +test("invalid dates, times and multi-line inline tables fail closed", (t) => { + const f = fixture(t); + for (const tail of [ + "foo = 2026-13-42", "foo = 25:99:99", "foo = [2026-99-99]", "foo = 2026-02-30", + "foo = 2026-01-01T10:00:00+24:00", "foo = { a = 1,\n b = 2 }", + ]) { + writeFileSync(join(f.codexHome, "config.toml"), FULL + tail + "\n"); + assert.equal(f.send(), "", tail); + assert.equal(f.advise(), "", tail); + } + for (const tail of ["foo = 2024-02-29", "foo = 1979-05-27T07:32:00Z", "foo = 07:32:00", "foo = { a = 1, b = \"x\" }"]) { + writeFileSync(join(f.codexHome, "config.toml"), FULL + tail + "\n"); + assert.equal(f.send(), ALLOW, tail); + } +}); + + test("network-access approval is allowed with opt-in", (t) => { const f = fixture(t); assert.equal(f.send({ tool_input: { description: "network-access example.com" } }), ALLOW); From 223d99b23a99f1a8498c6c635f0cda3b1492d80d Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 02:34:11 +0900 Subject: [PATCH 49/90] docs: publish measured test count (3679) --- README.ko.md | 2 +- README.md | 2 +- README.zh.md | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/README.ko.md b/README.ko.md index 6e1e10d4..189cb739 100644 --- a/README.ko.md +++ b/README.ko.md @@ -13,7 +13,7 @@

CI - 3,677 tests + 3,679 tests 29 skills 31 hooks Documentation diff --git a/README.md b/README.md index 84031047..ea74b530 100644 --- a/README.md +++ b/README.md @@ -13,7 +13,7 @@

CI - 3,677 tests + 3,679 tests 29 skills 31 hooks Documentation diff --git a/README.zh.md b/README.zh.md index d25ae253..1c163590 100644 --- a/README.zh.md +++ b/README.zh.md @@ -13,7 +13,7 @@

CI - 3,677 tests + 3,679 tests 29 skills 31 hooks Documentation From e7b87384b00c7c4026a009d455ce2d8a79dc80b2 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 02:38:10 +0900 Subject: [PATCH 50/90] fix(pabcd-state): overflowing numbers fail closed; record wp3 review --- .../020_wp3_agent_thread_permissions.md | 5 +++++ .../pabcd-state/dist/agent-thread-permissions.js | 9 ++++++++- .../pabcd-state/src/agent-thread-permissions.ts | 9 ++++++++- .../test/agent-thread-permissions.test.ts | 12 ++++++++++++ 4 files changed, 33 insertions(+), 2 deletions(-) diff --git a/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md b/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md index b85a120e..01afa30d 100644 --- a/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md +++ b/devlog/_plan/260927_issue_train/020_wp3_agent_thread_permissions.md @@ -321,3 +321,8 @@ Architect handle `01a0e3b8-436b-7203-a4f6-97d24b865814` re-checked this plan aft - **Whole-file TOML validation.** `codexConfigFullAccess` validates the entire file before trusting the two top-level keys; it no longer stops at the first header. A small structural scanner (no new dependency) walks the file: blank lines and `#` comments; table headers with paired delimiters; `key = value` statements where the value is a closed basic or literal string, a number, a boolean, an offset date-time, an inline table, or an array. Arrays and inline tables may span lines (the user's own config has `writable_roots = [` across lines); the scanner tracks bracket and brace depth outside strings and requires depth 0 at each statement end. Multi-line strings (`"""`, `'''`) are accepted when closed. Anything else, including an unclosed bracket at EOF, a stray token, or a duplicate top-level `approval_policy`/`sandbox_mode`, returns no decision. Tests: the user's pattern (keys, then `[features]`, then a multi-line `writable_roots` array) allows; `not_valid = [` at EOF after a valid table, `[[profiles]`, `[profiles]]`, `x = "unterminated` and `x = 1 2` get no decision. - **Canonical containment of the file actually read.** For both the default `~/.codexclaw/config.json` and a `CODEXCLAW_HOME` override, the handler resolves the real path of the config file it will read (`realpathSync`) and gives no decision when that real path lies inside the real cwd tree, with one exception: the unresolved default path itself (`/.codexclaw/config.json` when it is a regular file, not a symlink) stays eligible when cwd is the home directory, because that file is the user-global config by definition. A symlinked default file is judged by its target. Tests: override `/.codexclaw` (no decision), override outside cwd that is a symlink into `/.codexclaw` (no decision), outside `config.json` symlinked to a project file (no decision), plain outside override (allow), default home with cwd = home and a regular file (allow), default `~/.codexclaw/config.json` symlinked into the project (no decision). A test asserts `USER_ADVICE` mentions `agentCreatedThreadAutoAllow` and one-time network requests. - **Advice and scope text.** `MODEL_ADVICE` and the out-of-scope line now say that the opt-in answers one-time network requests and never changes the sandbox. The opt-in documentation states the same. + + +## wp3 implementation review record + +Reviewer `01a0e3dc-0b84-7db1-9166-938b6d14aa49` (gpt-6-sol, security focus) reviewed the built feature over four rounds. Folded: whole-file TOML grammar (headers, dotted keys, numbers, escapes, duplicates, repeated tables), canonical containment for both config files with a literal-default home exception, no hook observation writes for the two verbs, calendar-valid dates and times, single-line inline tables, finite numbers. Rebutted: Codex config *schema* type errors (for example `model = 123`) are not re-validated. The hook treats the two explicit top-level keys as evidence of user intent, not as a Codex schema check, and Codex refuses to start with a type-invalid config. This remains a documented residual risk alongside runtime overrides. diff --git a/plugins/codexclaw/components/pabcd-state/dist/agent-thread-permissions.js b/plugins/codexclaw/components/pabcd-state/dist/agent-thread-permissions.js index 03fda01b..e59b690e 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/agent-thread-permissions.js +++ b/plugins/codexclaw/components/pabcd-state/dist/agent-thread-permissions.js @@ -54,6 +54,13 @@ function boundedText(path ) { } /** Dates and times must name a real calendar day and clock time. */ +/** A decimal float or integer must be representable; `1e9999` overflows in Codex's parser. */ +function finiteNumber(token ) { + if (!/^[+-]?(?:\d|\.)/.test(token) || /^[+-]?(?:inf|nan)$/.test(token) || /^\d{4}-|^\d\d:/.test(token) || /^0[xob]/i.test(token)) return true; + const value = Number(token.replace(/_/g, "")); + return Number.isFinite(value) && (/[.eE]/.test(token) || Number.isSafeInteger(value) || Math.abs(value) <= 9223372036854775807); +} + function validDateTime(token ) { const date = /^(\d{4})-(\d\d)-(\d\d)/.exec(token); if (date) { @@ -289,7 +296,7 @@ function validValue(source ) { return false; } const bare = /^[^\s,\]\}#]+/.exec(source.slice(index)); - if (!bare || !scalar.test(bare[0]) || !validDateTime(bare[0])) return false; + if (!bare || !scalar.test(bare[0]) || !validDateTime(bare[0]) || !finiteNumber(bare[0])) return false; index += bare[0].length; return true; }; diff --git a/plugins/codexclaw/components/pabcd-state/src/agent-thread-permissions.ts b/plugins/codexclaw/components/pabcd-state/src/agent-thread-permissions.ts index 73611252..59c77c7a 100644 --- a/plugins/codexclaw/components/pabcd-state/src/agent-thread-permissions.ts +++ b/plugins/codexclaw/components/pabcd-state/src/agent-thread-permissions.ts @@ -54,6 +54,13 @@ function boundedText(path: string): string | null { } /** Dates and times must name a real calendar day and clock time. */ +/** A decimal float or integer must be representable; `1e9999` overflows in Codex's parser. */ +function finiteNumber(token: string): boolean { + if (!/^[+-]?(?:\d|\.)/.test(token) || /^[+-]?(?:inf|nan)$/.test(token) || /^\d{4}-|^\d\d:/.test(token) || /^0[xob]/i.test(token)) return true; + const value = Number(token.replace(/_/g, "")); + return Number.isFinite(value) && (/[.eE]/.test(token) || Number.isSafeInteger(value) || Math.abs(value) <= 9223372036854775807); +} + function validDateTime(token: string): boolean { const date = /^(\d{4})-(\d\d)-(\d\d)/.exec(token); if (date) { @@ -289,7 +296,7 @@ function validValue(source: string): boolean { return false; } const bare = /^[^\s,\]\}#]+/.exec(source.slice(index)); - if (!bare || !scalar.test(bare[0]) || !validDateTime(bare[0])) return false; + if (!bare || !scalar.test(bare[0]) || !validDateTime(bare[0]) || !finiteNumber(bare[0])) return false; index += bare[0].length; return true; }; diff --git a/plugins/codexclaw/components/pabcd-state/test/agent-thread-permissions.test.ts b/plugins/codexclaw/components/pabcd-state/test/agent-thread-permissions.test.ts index 6f7a0d8b..f792b23c 100644 --- a/plugins/codexclaw/components/pabcd-state/test/agent-thread-permissions.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/agent-thread-permissions.test.ts @@ -45,6 +45,18 @@ test("allows opted-in agent-created root thread for covered tools", (t) => { }); +test("overflowing numbers fail closed", (t) => { + const f = fixture(t); + for (const tail of ["foo = 1e9999", "foo = -1e400"]) { + writeFileSync(join(f.codexHome, "config.toml"), FULL + tail + "\n"); + assert.equal(f.send(), "", tail); + } + writeFileSync(join(f.codexHome, "config.toml"), FULL + "foo = 1e300\nbar = 9007199254740993\n"); + assert.equal(f.send(), ALLOW); +}); + + + test("symlinked default directories into the home-cwd project cannot grant permission", (t) => { const f = fixture(t); const env = { CODEX_HOME: "", CODEXCLAW_HOME: "", HOME: f.root }; From badaf5be123e635046f6677a5f02d9472e179b7a Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 02:38:10 +0900 Subject: [PATCH 51/90] docs: publish measured test count (3680) --- README.ko.md | 2 +- README.md | 2 +- README.zh.md | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/README.ko.md b/README.ko.md index 189cb739..82a189fe 100644 --- a/README.ko.md +++ b/README.ko.md @@ -13,7 +13,7 @@

CI - 3,679 tests + 3,680 tests 29 skills 31 hooks Documentation diff --git a/README.md b/README.md index ea74b530..5560778a 100644 --- a/README.md +++ b/README.md @@ -13,7 +13,7 @@

CI - 3,679 tests + 3,680 tests 29 skills 31 hooks Documentation diff --git a/README.zh.md b/README.zh.md index 1c163590..cd7ae04d 100644 --- a/README.zh.md +++ b/README.zh.md @@ -13,7 +13,7 @@

CI - 3,679 tests + 3,680 tests 29 skills 31 hooks Documentation From 3637fef1df572ea7d6fa55ecc3b277f4ad47b9a3 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 02:39:51 +0900 Subject: [PATCH 52/90] fix(pabcd-state): enforce signed 64-bit integers in the config scanner --- .../pabcd-state/dist/agent-thread-permissions.js | 14 +++++++++++--- .../pabcd-state/src/agent-thread-permissions.ts | 14 +++++++++++--- .../test/agent-thread-permissions.test.ts | 4 ++-- 3 files changed, 24 insertions(+), 8 deletions(-) diff --git a/plugins/codexclaw/components/pabcd-state/dist/agent-thread-permissions.js b/plugins/codexclaw/components/pabcd-state/dist/agent-thread-permissions.js index e59b690e..03af5778 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/agent-thread-permissions.js +++ b/plugins/codexclaw/components/pabcd-state/dist/agent-thread-permissions.js @@ -56,9 +56,17 @@ function boundedText(path ) { /** Dates and times must name a real calendar day and clock time. */ /** A decimal float or integer must be representable; `1e9999` overflows in Codex's parser. */ function finiteNumber(token ) { - if (!/^[+-]?(?:\d|\.)/.test(token) || /^[+-]?(?:inf|nan)$/.test(token) || /^\d{4}-|^\d\d:/.test(token) || /^0[xob]/i.test(token)) return true; - const value = Number(token.replace(/_/g, "")); - return Number.isFinite(value) && (/[.eE]/.test(token) || Number.isSafeInteger(value) || Math.abs(value) <= 9223372036854775807); + if (!/^[+-]?(?:\d|\.)/.test(token) || /^[+-]?(?:inf|nan)$/.test(token) || /^\d{4}-|^\d\d:/.test(token)) return true; + const clean = token.replace(/_/g, ""); + const radix = /^0[xob]/i.exec(clean); + if (radix || !/[.eE]/.test(clean)) { + // Integers of every radix must fit a signed 64-bit value, as in Codex's parser. + const negative = clean.startsWith("-"); + const digits = clean.replace(/^[+-]/, ""); + const value = BigInt(radix ? `0${digits[1].toLowerCase()}${digits.slice(2)}` : digits); + return negative ? value <= 9223372036854775808n : value <= 9223372036854775807n; + } + return Number.isFinite(Number(clean)); } function validDateTime(token ) { diff --git a/plugins/codexclaw/components/pabcd-state/src/agent-thread-permissions.ts b/plugins/codexclaw/components/pabcd-state/src/agent-thread-permissions.ts index 59c77c7a..86e4dc0d 100644 --- a/plugins/codexclaw/components/pabcd-state/src/agent-thread-permissions.ts +++ b/plugins/codexclaw/components/pabcd-state/src/agent-thread-permissions.ts @@ -56,9 +56,17 @@ function boundedText(path: string): string | null { /** Dates and times must name a real calendar day and clock time. */ /** A decimal float or integer must be representable; `1e9999` overflows in Codex's parser. */ function finiteNumber(token: string): boolean { - if (!/^[+-]?(?:\d|\.)/.test(token) || /^[+-]?(?:inf|nan)$/.test(token) || /^\d{4}-|^\d\d:/.test(token) || /^0[xob]/i.test(token)) return true; - const value = Number(token.replace(/_/g, "")); - return Number.isFinite(value) && (/[.eE]/.test(token) || Number.isSafeInteger(value) || Math.abs(value) <= 9223372036854775807); + if (!/^[+-]?(?:\d|\.)/.test(token) || /^[+-]?(?:inf|nan)$/.test(token) || /^\d{4}-|^\d\d:/.test(token)) return true; + const clean = token.replace(/_/g, ""); + const radix = /^0[xob]/i.exec(clean); + if (radix || !/[.eE]/.test(clean)) { + // Integers of every radix must fit a signed 64-bit value, as in Codex's parser. + const negative = clean.startsWith("-"); + const digits = clean.replace(/^[+-]/, ""); + const value = BigInt(radix ? `0${digits[1].toLowerCase()}${digits.slice(2)}` : digits); + return negative ? value <= 9223372036854775808n : value <= 9223372036854775807n; + } + return Number.isFinite(Number(clean)); } function validDateTime(token: string): boolean { diff --git a/plugins/codexclaw/components/pabcd-state/test/agent-thread-permissions.test.ts b/plugins/codexclaw/components/pabcd-state/test/agent-thread-permissions.test.ts index f792b23c..dac8524b 100644 --- a/plugins/codexclaw/components/pabcd-state/test/agent-thread-permissions.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/agent-thread-permissions.test.ts @@ -47,11 +47,11 @@ test("allows opted-in agent-created root thread for covered tools", (t) => { test("overflowing numbers fail closed", (t) => { const f = fixture(t); - for (const tail of ["foo = 1e9999", "foo = -1e400"]) { + for (const tail of ["foo = 1e9999", "foo = -1e400", "foo = 9223372036854775808", "foo = 0x8000000000000000", "foo = -9223372036854775809"]) { writeFileSync(join(f.codexHome, "config.toml"), FULL + tail + "\n"); assert.equal(f.send(), "", tail); } - writeFileSync(join(f.codexHome, "config.toml"), FULL + "foo = 1e300\nbar = 9007199254740993\n"); + writeFileSync(join(f.codexHome, "config.toml"), FULL + "foo = 1e300\nbar = 9007199254740993\nbaz = 9223372036854775807\nqux = -9223372036854775808\nhex = 0x7fff_ffff_ffff_ffff\n"); assert.equal(f.send(), ALLOW); }); From a5bb1305a2e1fe375b86ab525316d90ca96fcb49 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 02:49:44 +0900 Subject: [PATCH 53/90] docs(plan): re-anchor wp4 and add the all-waiting IDLE Stop release --- .../_plan/260927_issue_train/030_wp4_goalplan_decisions.md | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/devlog/_plan/260927_issue_train/030_wp4_goalplan_decisions.md b/devlog/_plan/260927_issue_train/030_wp4_goalplan_decisions.md index 5e466bc8..567bd24b 100644 --- a/devlog/_plan/260927_issue_train/030_wp4_goalplan_decisions.md +++ b/devlog/_plan/260927_issue_train/030_wp4_goalplan_decisions.md @@ -302,3 +302,9 @@ These tests must include the negative enforcement paths because a filtered `read ## Out of scope No host question tool integration, automatic answer capture, reminder/polling mechanism, `options[]` validation, `withdrawn` status, global goal pause, schema migration/version bump, Interview ledger reuse, or new standalone decision file. Issue #262's historical option/recommendation proposal differs from this phase's delegated free-text contract; if the parent wants the issue's options semantics, it needs a separate contract decision before implementation. + + +## wp4 re-verification (architect 01a0e3fa-8b59-75b0-a7a4-d902727608f9, supersedes stale text above) + +- **Provenance.** Anchors re-checked on `codex/issue-train-wp3` at 3637fef1 (wp2 and wp3 landed; goalplan files unchanged by them). Current line numbers: reviver ends at goalplan.ts:607; `withGoalplanWriteLock` hands the parsed plan to its callback (goalplan.ts:741/797), so `runDecision` uses that argument instead of re-reading; readiness entries goalplan.ts:986, 993-1007, 1053-1120; ID regex :1123; integrity :1314; remaining-work validation :1494-1496; close/successor :1798, :1825-1829, :1872-1879; absent-target wording :1921-1928 and gate :1957; cursor :2036-2053. +- **Stop at IDLE when only decisions remain (W4-4).** Line 205's "no hook-specific filter is needed" holds for target selection but not for the IDLE continuation block (`hook.ts:1823-1831`), which blocks every active, bound IDLE goal. Add `remainingWorkAwaitsDecisions(plan): boolean` to goalplan.ts: true when at least one work phase is not done, no work phase is in progress or runnable (`isRunnablePhase`), and at least one not-done phase lists an open decision in `awaitsDecision`. In `handleStop`, after the bound-plan check and before the counter, `if (remainingWorkAwaitsDecisions(plan)) return "";` (no counter write). The goal stays active and GOAL-COMPLETE-GATE-01 still refuses completion. Tests in `hook-continuation.test.ts`: `IDLE Stop releases when every remaining phase awaits an open decision` (no block, stopBlockTotal unchanged); `IDLE Stop still blocks when an independent phase is runnable` (one waiting phase plus one ready phase); after `decide`, the same plan blocks again with the arming command. Also unit tests for the helper in the decisions test file. From 5ef2ca5173710df4ae6bcba08df1f01c5fe8184a Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 02:50:17 +0900 Subject: [PATCH 54/90] docs(plan): wp4 IDLE release requires every remaining phase to wait on a decision --- devlog/_plan/260927_issue_train/030_wp4_goalplan_decisions.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/devlog/_plan/260927_issue_train/030_wp4_goalplan_decisions.md b/devlog/_plan/260927_issue_train/030_wp4_goalplan_decisions.md index 567bd24b..f6877fd8 100644 --- a/devlog/_plan/260927_issue_train/030_wp4_goalplan_decisions.md +++ b/devlog/_plan/260927_issue_train/030_wp4_goalplan_decisions.md @@ -307,4 +307,4 @@ No host question tool integration, automatic answer capture, reminder/polling me ## wp4 re-verification (architect 01a0e3fa-8b59-75b0-a7a4-d902727608f9, supersedes stale text above) - **Provenance.** Anchors re-checked on `codex/issue-train-wp3` at 3637fef1 (wp2 and wp3 landed; goalplan files unchanged by them). Current line numbers: reviver ends at goalplan.ts:607; `withGoalplanWriteLock` hands the parsed plan to its callback (goalplan.ts:741/797), so `runDecision` uses that argument instead of re-reading; readiness entries goalplan.ts:986, 993-1007, 1053-1120; ID regex :1123; integrity :1314; remaining-work validation :1494-1496; close/successor :1798, :1825-1829, :1872-1879; absent-target wording :1921-1928 and gate :1957; cursor :2036-2053. -- **Stop at IDLE when only decisions remain (W4-4).** Line 205's "no hook-specific filter is needed" holds for target selection but not for the IDLE continuation block (`hook.ts:1823-1831`), which blocks every active, bound IDLE goal. Add `remainingWorkAwaitsDecisions(plan): boolean` to goalplan.ts: true when at least one work phase is not done, no work phase is in progress or runnable (`isRunnablePhase`), and at least one not-done phase lists an open decision in `awaitsDecision`. In `handleStop`, after the bound-plan check and before the counter, `if (remainingWorkAwaitsDecisions(plan)) return "";` (no counter write). The goal stays active and GOAL-COMPLETE-GATE-01 still refuses completion. Tests in `hook-continuation.test.ts`: `IDLE Stop releases when every remaining phase awaits an open decision` (no block, stopBlockTotal unchanged); `IDLE Stop still blocks when an independent phase is runnable` (one waiting phase plus one ready phase); after `decide`, the same plan blocks again with the arming command. Also unit tests for the helper in the decisions test file. +- **Stop at IDLE when only decisions remain (W4-4).** Line 205's "no hook-specific filter is needed" holds for target selection but not for the IDLE continuation block (`hook.ts:1823-1831`), which blocks every active, bound IDLE goal. Add `remainingWorkAwaitsDecisions(plan): boolean` to goalplan.ts: true when at least one work phase is not done and **every** not-done phase (pending or in progress) is decision-waiting. A phase is decision-waiting when it lists an open decision in `awaitsDecision`, or when some prerequisite in its `dependsOn` closure that is not done is itself decision-waiting (computed with a visited set, so cycles return false). A phase that is `blocked` for another reason, runnable, or waiting only on non-decision prerequisites makes the helper false and keeps today's continuation path. In `handleStop`, after the bound-plan check and before the counter, `if (remainingWorkAwaitsDecisions(plan)) return "";` (no counter write). The goal stays active and GOAL-COMPLETE-GATE-01 still refuses completion. Tests in `hook-continuation.test.ts`: `IDLE Stop releases when every remaining phase awaits an open decision` (no block, stopBlockTotal unchanged); `IDLE Stop still blocks when an independent phase is runnable` (one waiting phase plus one ready phase); after `decide`, the same plan blocks again with the arming command; `IDLE Stop still blocks when one phase waits on a decision and another is blocked for another reason`; `IDLE Stop releases when an in-progress phase gained an open decision mid-cycle and its dependents wait on it`. Also unit tests for the helper in the decisions test file. From e65b14efdf85b30addddb52639eeda068df928f5 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 02:54:39 +0900 Subject: [PATCH 55/90] docs(plan): fold wp4 audit round 1 --- .../030_wp4_goalplan_decisions.md | 15 +++++++++++++-- 1 file changed, 13 insertions(+), 2 deletions(-) diff --git a/devlog/_plan/260927_issue_train/030_wp4_goalplan_decisions.md b/devlog/_plan/260927_issue_train/030_wp4_goalplan_decisions.md index f6877fd8..fed12fa4 100644 --- a/devlog/_plan/260927_issue_train/030_wp4_goalplan_decisions.md +++ b/devlog/_plan/260927_issue_train/030_wp4_goalplan_decisions.md @@ -82,8 +82,12 @@ Add these pure helpers beside `isRunnablePhase` at `:983` and replace its body. ```ts export function openDecisionIdsForPhase(plan: Goalplan, wp: GoalplanWorkPhase): string[] { - return [...new Set(wp.awaitsDecision ?? [])].filter((id) => - plan.decisions?.find((decision) => decision.id === id)?.status !== "decided"); + // A reference counts as answered only when exactly one decision has that id and it + // is decided; a missing or duplicated id keeps the phase waiting (fail closed). + return [...new Set(wp.awaitsDecision ?? [])].filter((id) => { + const matches = (plan.decisions ?? []).filter((decision) => decision.id === id); + return !(matches.length === 1 && matches[0].status === "decided"); + }); } function workPhaseReadyConditionsMet(plan: Goalplan, wp: GoalplanWorkPhase): boolean { @@ -308,3 +312,10 @@ No host question tool integration, automatic answer capture, reminder/polling me - **Provenance.** Anchors re-checked on `codex/issue-train-wp3` at 3637fef1 (wp2 and wp3 landed; goalplan files unchanged by them). Current line numbers: reviver ends at goalplan.ts:607; `withGoalplanWriteLock` hands the parsed plan to its callback (goalplan.ts:741/797), so `runDecision` uses that argument instead of re-reading; readiness entries goalplan.ts:986, 993-1007, 1053-1120; ID regex :1123; integrity :1314; remaining-work validation :1494-1496; close/successor :1798, :1825-1829, :1872-1879; absent-target wording :1921-1928 and gate :1957; cursor :2036-2053. - **Stop at IDLE when only decisions remain (W4-4).** Line 205's "no hook-specific filter is needed" holds for target selection but not for the IDLE continuation block (`hook.ts:1823-1831`), which blocks every active, bound IDLE goal. Add `remainingWorkAwaitsDecisions(plan): boolean` to goalplan.ts: true when at least one work phase is not done and **every** not-done phase (pending or in progress) is decision-waiting. A phase is decision-waiting when it lists an open decision in `awaitsDecision`, or when some prerequisite in its `dependsOn` closure that is not done is itself decision-waiting (computed with a visited set, so cycles return false). A phase that is `blocked` for another reason, runnable, or waiting only on non-decision prerequisites makes the helper false and keeps today's continuation path. In `handleStop`, after the bound-plan check and before the counter, `if (remainingWorkAwaitsDecisions(plan)) return "";` (no counter write). The goal stays active and GOAL-COMPLETE-GATE-01 still refuses completion. Tests in `hook-continuation.test.ts`: `IDLE Stop releases when every remaining phase awaits an open decision` (no block, stopBlockTotal unchanged); `IDLE Stop still blocks when an independent phase is runnable` (one waiting phase plus one ready phase); after `decide`, the same plan blocks again with the arming command; `IDLE Stop still blocks when one phase waits on a decision and another is blocked for another reason`; `IDLE Stop releases when an in-progress phase gained an open decision mid-cycle and its dependents wait on it`. Also unit tests for the helper in the decisions test file. + + +## wp4 audit folds (round 1) + +- **Duplicate or missing decision ids fail closed.** `openDecisionIdsForPhase` (code above, edited in place) treats a reference as answered only when exactly one decision has that id and it is decided. Tests: `duplicate decision id keeps the phase waiting` for both orders (decided then open, open then decided) through `effectiveActiveWorkPhaseId` cursor selection, `closeFixedWorkPhase` successor choice, and absent-target recovery; `missing decision id keeps the phase waiting`. +- **Unmet criteria gate the IDLE release.** `remainingWorkAwaitsDecisions(plan)` additionally requires every unmet criterion to be listed in the `criteriaIds` of some decision-waiting phase. An unmet criterion that no phase lists, or that only a non-waiting phase lists, keeps today's continuation block, because the agent may still be able to act on it. Tests: `IDLE Stop still blocks when an independent criterion is unmet` and `IDLE Stop releases when every unmet criterion belongs to a decision-waiting phase`. +- **Bypass record for the Stop release.** Tier: E4 Stop continuation (a release decision, not a denial). Executing surface: `handleStop` IDLE branch via `remainingWorkAwaitsDecisions`. Known bypass: hand-editing the goalplan to add `awaitsDecision` links or an open decision releases Stop early; `cxc loop ask` records a question without proving it was delivered. Residual risk: an agent can park a goal by recording a decision it never asked; the goal stays active and GOAL-COMPLETE-GATE-01 still refuses completion, so the work cannot be falsely closed, only paused. Wording: "Stop releases at IDLE when every remaining phase and unmet criterion waits on an open user decision." Final gate: GOAL-COMPLETE-GATE-01 via E8 validation, unchanged. From 40624cdb9d1fbf3b0c4c6fecd184c26de16551c8 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 02:56:19 +0900 Subject: [PATCH 56/90] test(pabcd-state): set USERPROFILE with HOME so Windows homedir() sees the fixture home --- .../pabcd-state/test/agent-thread-permissions.test.ts | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/plugins/codexclaw/components/pabcd-state/test/agent-thread-permissions.test.ts b/plugins/codexclaw/components/pabcd-state/test/agent-thread-permissions.test.ts index dac8524b..6651fcdb 100644 --- a/plugins/codexclaw/components/pabcd-state/test/agent-thread-permissions.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/agent-thread-permissions.test.ts @@ -59,7 +59,7 @@ test("overflowing numbers fail closed", (t) => { test("symlinked default directories into the home-cwd project cannot grant permission", (t) => { const f = fixture(t); - const env = { CODEX_HOME: "", CODEXCLAW_HOME: "", HOME: f.root }; + const env = { CODEX_HOME: "", CODEXCLAW_HOME: "", HOME: f.root, USERPROFILE: f.root }; // Real project directories under the home, reached through symlinked defaults. const projectCodex = join(f.cwd, ".codex"); const projectClaw = join(f.cwd, ".codexclaw"); @@ -302,7 +302,7 @@ test("default home config remains eligible at home but a project symlink does no const codexDir = join(f.root, ".codex"); mkdirSync(codexDir); writeFileSync(join(codexDir, "config.toml"), FULL); - const env = { CODEX_HOME: "", CODEXCLAW_HOME: "", HOME: f.root }; + const env = { CODEX_HOME: "", CODEXCLAW_HOME: "", HOME: f.root, USERPROFILE: f.root }; const atHome = cli("permission-request", { ...f.input, cwd: f.root }, f.root, env); assert.equal(atHome.status, 0, atHome.stderr); assert.equal(atHome.stdout, ALLOW); From 340cc542165ae478280d4dccd180acd7a7492c53 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 03:14:51 +0900 Subject: [PATCH 57/90] Add plan-local decision waits and IDLE Stop release --- .../pabcd-state/src/goalplan-cli.ts | 94 ++++++++- .../components/pabcd-state/src/goalplan.ts | 191 ++++++++++++++++-- .../components/pabcd-state/src/hook.ts | 5 +- .../test/goalplan-public-surface.test.ts | 100 ++++++++- .../test/hook-continuation.test.ts | 82 ++++++++ .../pabcd-state/test/orchestrate-cli.test.ts | 2 +- .../test/work-phase-states.test.ts | 121 +++++++++++ 7 files changed, 572 insertions(+), 23 deletions(-) diff --git a/plugins/codexclaw/components/pabcd-state/src/goalplan-cli.ts b/plugins/codexclaw/components/pabcd-state/src/goalplan-cli.ts index 78dde175..e6d584dc 100644 --- a/plugins/codexclaw/components/pabcd-state/src/goalplan-cli.ts +++ b/plugins/codexclaw/components/pabcd-state/src/goalplan-cli.ts @@ -13,6 +13,9 @@ */ import { addGoalplanTask, + askGoalplanDecision, + decideGoalplanDecision, + openDecisionIdsForPhase, buildGoalplan, completeGoalplanTask, goalplanDefinitionIntegrityReasons, @@ -59,6 +62,8 @@ export type GoalplanVerb = | "add-task" | "complete-task" | "meet-criterion" + | "ask" + | "decide" | "help"; export interface GoalplanCliArgs { @@ -98,6 +103,10 @@ export interface GoalplanCliArgs { dependsOn?: string[]; /** `add-task` / `complete-task`: which work phase owns the task. */ workPhaseId?: string; + workPhaseIds?: string[]; + question?: string; + recommendation?: string; + answer?: string; /** `complete-task`: the outcome evidence a done task must carry. */ outcome?: string; /** `meet-criterion`: the captured evidence a met criterion must carry. */ @@ -124,12 +133,15 @@ const VERBS: ReadonlySet = new Set([ "add-task", "complete-task", "meet-criterion", + "ask", + "decide", ]); type GoalplanFlag = | "--objective" | "--slug" | "--criterion" | "--cwd" | "--session" | "--batch-json" | "--surface" | "--presented" | "--id" | "--title" | "--work-phase" - | "--outcome" | "--schema-version" | "--evidence" | "--json" | "--depends-on"; + | "--outcome" | "--schema-version" | "--evidence" | "--json" | "--depends-on" + | "--question" | "--recommendation" | "--answer"; type VerbRule = { allowed: ReadonlySet; @@ -148,6 +160,8 @@ const VERB_RULES: Readonly> = { "add-task": { allowed: new Set(["--session", "--work-phase", "--id", "--title", "--depends-on", "--cwd"]), repeatable: new Set(["--depends-on"]), usage: "add-task --session --work-phase --id --title [--depends-on ]... [--cwd ]" }, "complete-task": { allowed: new Set(["--session", "--work-phase", "--id", "--outcome", "--cwd"]), repeatable: new Set(), usage: "complete-task --session --work-phase --id --outcome [--cwd ]" }, "meet-criterion": { allowed: new Set(["--session", "--id", "--evidence", "--cwd"]), repeatable: new Set(), usage: "meet-criterion --session --id --evidence [--cwd ]" }, + ask: { allowed: new Set(["--session", "--id", "--question", "--recommendation", "--work-phase", "--cwd"]), repeatable: new Set(["--work-phase"]), usage: "ask --session --id --question [--recommendation ] [--work-phase ]... [--cwd ]" }, + decide: { allowed: new Set(["--session", "--id", "--answer", "--cwd"]), repeatable: new Set(), usage: "decide --session --id --answer [--cwd ]" }, help: { allowed: new Set(), repeatable: new Set(), usage: "--help" }, }; @@ -163,12 +177,12 @@ export function parseGoalplanCliArgs(argv: string[], cwd: string): GoalplanCliAr } if (!VERBS.has(verb)) { return { - error: `unknown loop verb '${argv[0] ?? ""}' (expected init|show|validate|steer|add-criterion|add-work-phase|ready|add-task|complete-task|meet-criterion); run cxc loop --help`, + error: `unknown loop verb '${argv[0] ?? ""}' (expected init|show|validate|steer|add-criterion|add-work-phase|ready|add-task|complete-task|meet-criterion|ask|decide); run cxc loop --help`, }; } const selected = verb as GoalplanVerb; const rule = VERB_RULES[selected]; - const out: GoalplanCliArgs = { verb: selected, cwd, criteria: [], dependsOn: [] }; + const out: GoalplanCliArgs = { verb: selected, cwd, criteria: [], dependsOn: [], workPhaseIds: [] }; const seen = new Set(); const reject = (message: string): GoalplanCliParseError => ({ error: `${selected}: ${message}` }); for (let i = 1; i < argv.length; i++) { @@ -213,7 +227,17 @@ export function parseGoalplanCliArgs(argv: string[], cwd: string): GoalplanCliAr case "--presented": out.presented = value; break; case "--id": out.id = value; break; case "--title": out.title = value; break; - case "--work-phase": out.workPhaseId = value; break; + case "--work-phase": { + if (selected !== "ask") { out.workPhaseId = value; break; } + const phaseId = value.trim(); + if (!phaseId) return reject("--work-phase requires one non-empty id"); + if (out.workPhaseIds!.includes(phaseId)) return reject(`--work-phase must not repeat id '${phaseId}'`); + out.workPhaseIds!.push(phaseId); + break; + } + case "--question": out.question = value; break; + case "--recommendation": out.recommendation = value; break; + case "--answer": out.answer = value; break; case "--outcome": out.outcome = value; break; case "--schema-version": { const parsed = Number(value); @@ -438,6 +462,13 @@ function runReady(args: GoalplanCliArgs, plan: Goalplan): GoalplanCliResult { const phases = readyWorkPhases(plan); const tasks = readyTasks(plan); + const openDecisions = (plan.decisions ?? []).filter((decision) => decision.status === "open") + .map(({ id, question, recommendation, askedAt }) => ({ id, question, ...(recommendation === undefined ? {} : { recommendation }), askedAt })); + const awaitingDecisions = plan.workPhases + .filter((wp) => wp.status === "pending" || wp.status === "in_progress") + .map((wp) => ({ workPhaseId: wp.id, decisionIds: openDecisionIdsForPhase(plan, wp).filter((id) => + (plan.decisions ?? []).some((decision) => decision.id === id && decision.status === "open")) })) + .filter((entry) => entry.decisionIds.length > 0); if (args.json === true) { return { output: JSON.stringify({ @@ -455,6 +486,7 @@ function runReady(args: GoalplanCliArgs, plan: Goalplan): GoalplanCliResult { id: entry.task.id, title: entry.task.title, })), + ...(plan.decisions === undefined ? {} : { openDecisions, awaitingDecisions }), }), code: 0, }; @@ -467,9 +499,52 @@ function runReady(args: GoalplanCliArgs, plan: Goalplan): GoalplanCliResult { lines.push(tasks.length > 0 ? `readyTasks: ${tasks.map((entry) => `${entry.workPhaseId}/${entry.task.id} (${entry.task.title})`).join("; ")}` : "readyTasks: none"); + if (plan.decisions !== undefined) { + lines.push(openDecisions.length > 0 + ? `openDecisions: ${openDecisions.map((decision) => `${decision.id} (${decision.question})`).join("; ")}` + : "openDecisions: none"); + lines.push(awaitingDecisions.length > 0 + ? `awaitingDecisions: ${awaitingDecisions.map((entry) => `${entry.workPhaseId}: ${entry.decisionIds.join(", ")}`).join("; ")}` + : "awaitingDecisions: none"); + } return { output: lines.join("\n"), code: 0 }; } +/** Record a host-submitted question or its answer under the goalplan write lock. */ +function runDecision(args: GoalplanCliArgs): GoalplanCliResult { + const session = (args.session ?? "").trim(); + if (!session) return { output: `loop ${args.verb}: --session is required`, code: 1 }; + if (!isCanonicalSessionId(session)) return { output: `loop ${args.verb}: session id is not canonical`, code: 1 }; + const slug = readState(args.cwd, session).slug; + if (!slug) return { output: `loop ${args.verb}: session '${session}' has no bound goalplan - run \`cxc loop init --session ${session}\` first`, code: 1 }; + const id = (args.id ?? "").trim(); + if (args.verb === "ask" && (!id || !(args.question ?? "").trim())) { + return { output: "loop ask: --id and non-empty --question are required", code: 1 }; + } + if (args.verb === "decide" && (!id || !(args.answer ?? "").trim())) { + return { output: "loop decide: --id and non-empty --answer are required", code: 1 }; + } + type DecisionCommit = { kind: "rejected"; reason: string } | { kind: "changed" } | { kind: "unchanged"; reason: string }; + const locked = withGoalplanWriteLock(args.cwd, slug, (plan) => { + const result = args.verb === "ask" + ? askGoalplanDecision(plan, { + id, question: args.question!, recommendation: args.recommendation, + workPhaseIds: args.workPhaseIds ?? [], askedAt: new Date().toISOString(), + }) + : decideGoalplanDecision(plan, id, args.answer!, new Date().toISOString()); + if (result.kind === "rejected") return { kind: "rejected", reason: result.reason }; + if (result.kind === "unchanged") return { kind: "unchanged", reason: result.reason }; + writeGoalplan(args.cwd, result.plan); + return { kind: "changed" }; + }); + if (locked.kind === "locked" || locked.kind === "unreadable") { + return { output: `loop ${args.verb}: ${locked.reason}`, code: 1 }; + } + if (locked.value.kind === "rejected") return { output: `loop ${args.verb}: ${locked.value.reason}`, code: 1 }; + if (locked.value.kind === "unchanged") return { output: `loop ${args.verb}: ${locked.value.reason}; nothing to do`, code: 0 }; + return { output: `loop ${args.verb}: ${slug} ${id} applied`, code: 0 }; +} + /** * 060 wp6: the three lifecycle verbs share one locked read-modify-write. * @@ -610,6 +685,12 @@ function renderPlanLines(plan: Goalplan, lock?: GoalplanWriteLockStatus): string for (const c of plan.criteria) { lines.push(` - ${c.id} [${c.status}] ${c.scenario}`); } + for (const decision of plan.decisions ?? []) { + if (decision.status !== "open") continue; + lines.push(` - ${decision.id} [open] ${decision.question}`); + const waiting = plan.workPhases.filter((wp) => wp.awaitsDecision?.includes(decision.id)); + lines.push(` waiting: ${waiting.map((wp) => wp.id).join(", ") || "none"}`); + } return lines.join("\n"); } @@ -623,7 +704,7 @@ export function renderGoalplanHelp(): string { "cxc loop — durable goalplan for a multi-cycle PABCD loop", "", "Usage:", - ...(["init", "show", "validate", "steer", "add-criterion", "add-work-phase", "ready", "add-task", "complete-task", "meet-criterion", "help"] as const) + ...(["init", "show", "validate", "steer", "add-criterion", "add-work-phase", "ready", "add-task", "complete-task", "meet-criterion", "ask", "decide", "help"] as const) .map((verb) => ` cxc loop ${VERB_RULES[verb].usage}`), "", "Notes:", @@ -639,6 +720,8 @@ export function renderGoalplanHelp(): string { " additionally require an approved finalGate, and no verb in this build opens a", " final-gate review round, so opt in only if you can record that gate yourself.", " meet-criterion requires non-empty captured evidence for the same reason.", + " Send the question through the host first, then record it with ask; ask never sends a message.", + " Record the user's reply with decide. It changes only the decision record.", "", "steer --batch-json expects an object with:", ' { "idempotencyKey": "", "rationale": "", "evidence": "",', @@ -710,6 +793,7 @@ export function runGoalplanCli(args: GoalplanCliArgs): GoalplanCliResult { } if (args.verb === "steer") return runSteer(args); + if (args.verb === "ask" || args.verb === "decide") return runDecision(args); if (args.verb === "add-criterion" || args.verb === "add-work-phase") return runAddOp(args); diff --git a/plugins/codexclaw/components/pabcd-state/src/goalplan.ts b/plugins/codexclaw/components/pabcd-state/src/goalplan.ts index 7fd75cc7..4bc77b96 100644 --- a/plugins/codexclaw/components/pabcd-state/src/goalplan.ts +++ b/plugins/codexclaw/components/pabcd-state/src/goalplan.ts @@ -130,12 +130,24 @@ export interface GoalplanWorkPhase { criteriaIds: string[]; /** Work-phase ids in this plan that must be `done` first; see GoalplanTask.dependsOn. */ dependsOn?: string[]; + /** Decision ids; any open target pauses this phase without changing status. */ + awaitsDecision?: string[]; /** why a `blocked` phase cannot proceed; cleared when it is unblocked. */ blockedReason?: string; /** the phase that took over the work; required on a `superseded` phase. */ supersededBy?: string; } +export interface GoalplanDecision { + id: string; + question: string; + recommendation?: string; + status: "open" | "decided"; + answer?: string; + askedAt: string; + decidedAt?: string; +} + export interface GoalplanHostLink { /** true only after a freeze-boundary arm (the MAIN session created a goal). */ armed: boolean; @@ -222,6 +234,7 @@ export interface Goalplan { /** the durable work-phase cursor the FSM does NOT hold across a D-close. */ activeWorkPhaseId: string | null; workPhases: GoalplanWorkPhase[]; + decisions?: GoalplanDecision[]; criteria: GoalplanCriterion[]; host: GoalplanHostLink; /** review rounds, oldest first. Absent on plans created before 010. */ @@ -504,6 +517,37 @@ function reviveDependsOn(value: unknown): string[] | undefined | "invalid" { return ids; } +function validIsoTime(value: unknown): value is string { + if (typeof value !== "string") return false; + const date = new Date(value); + return Number.isFinite(date.valueOf()) && date.toISOString() === value; +} + +function reviveDecisions(value: unknown): GoalplanDecision[] | undefined | "invalid" { + if (value === undefined) return undefined; + if (!Array.isArray(value)) return "invalid"; + const decisions: GoalplanDecision[] = []; + for (const item of value) { + if (typeof item !== "object" || item === null || Array.isArray(item)) return "invalid"; + const d = item as Record; + if (typeof d.id !== "string" || !LIFECYCLE_ID_RE.test(d.id) + || typeof d.question !== "string" || !d.question.trim() + || !validIsoTime(d.askedAt) + || (d.recommendation !== undefined && (typeof d.recommendation !== "string" || !d.recommendation.trim()))) return "invalid"; + if (d.status === "open") { + if (d.answer !== undefined || d.decidedAt !== undefined) return "invalid"; + decisions.push({ id: d.id, question: d.question, status: "open", askedAt: d.askedAt, + ...(d.recommendation === undefined ? {} : { recommendation: d.recommendation as string }) }); + } else if (d.status === "decided") { + if (typeof d.answer !== "string" || !d.answer.trim() || !validIsoTime(d.decidedAt)) return "invalid"; + decisions.push({ id: d.id, question: d.question, status: "decided", answer: d.answer, + askedAt: d.askedAt, decidedAt: d.decidedAt, + ...(d.recommendation === undefined ? {} : { recommendation: d.recommendation as string }) }); + } else return "invalid"; + } + return decisions; +} + /** Best-effort structural validation; a malformed object reads as absent (null). */ function reviveGoalplan(parsed: unknown, expectedSlug?: string): Goalplan | null { if (typeof parsed !== "object" || parsed === null || Array.isArray(parsed)) return null; @@ -525,6 +569,8 @@ function reviveGoalplan(parsed: unknown, expectedSlug?: string): Goalplan | null if (typeof w.id !== "string" || typeof w.title !== "string") return null; const phaseDependsOn = reviveDependsOn(w.dependsOn); if (phaseDependsOn === "invalid") return null; + const awaitsDecision = reviveDependsOn(w.awaitsDecision); + if (awaitsDecision === "invalid") return null; const status: WorkPhaseStatus = w.status === "in_progress" || w.status === "done" || w.status === "blocked" || w.status === "superseded" ? w.status @@ -548,6 +594,7 @@ function reviveGoalplan(parsed: unknown, expectedSlug?: string): Goalplan | null : []; const phase: GoalplanWorkPhase = { id: w.id, title: w.title, status, tasks, criteriaIds }; if (phaseDependsOn !== undefined) phase.dependsOn = phaseDependsOn; + if (awaitsDecision !== undefined) phase.awaitsDecision = awaitsDecision; if (typeof w.blockedReason === "string") phase.blockedReason = w.blockedReason; if (typeof w.supersededBy === "string") phase.supersededBy = w.supersededBy; workPhases.push(phase); @@ -580,6 +627,8 @@ function reviveGoalplan(parsed: unknown, expectedSlug?: string): Goalplan | null }; const reviewRounds = reviveReviewRounds(o.reviewRounds); + const decisions = reviveDecisions(o.decisions); + if (decisions === "invalid") return null; const plan: Goalplan = { objective: o.objective, @@ -594,6 +643,7 @@ function reviveGoalplan(parsed: unknown, expectedSlug?: string): Goalplan | null // Only attach the 010 fields when they are actually present, so a plan written // before this feature round-trips byte-identical. if (reviewRounds !== undefined) plan.reviewRounds = reviewRounds; + if (decisions !== undefined) plan.decisions = decisions; if (typeof o.activePlanAuditRoundId === "string") plan.activePlanAuditRoundId = o.activePlanAuditRoundId; if (typeof o.activeFinalGateRoundId === "string") plan.activeFinalGateRoundId = o.activeFinalGateRoundId; if (typeof o.schemaVersion === "number" && Number.isFinite(o.schemaVersion)) { @@ -822,6 +872,7 @@ function firstInvalidField(parsed: unknown): string { for (const rawWp of o.workPhases) { const wp = rawWp as Record; if (reviveDependsOn(wp.dependsOn) === "invalid") return "workPhases[].dependsOn"; + if (reviveDependsOn(wp.awaitsDecision) === "invalid") return "workPhases[].awaitsDecision"; for (const rawTask of Array.isArray(wp.tasks) ? wp.tasks : []) { if (typeof rawTask !== "object" || rawTask === null) continue; const task = rawTask as Record; @@ -839,6 +890,7 @@ function firstInvalidField(parsed: unknown): string { if (typeof o.host !== "object" || o.host === null || typeof (o.host as Record).armed !== "boolean") { return "host (needs armed/armedAt/source)"; } + if (reviveDecisions(o.decisions) === "invalid") return "decisions"; if (o.steeringLog !== undefined && !Array.isArray(o.steeringLog)) return "steeringLog"; return "(unknown)"; } @@ -983,11 +1035,21 @@ function taskDependenciesMet(phase: GoalplanWorkPhase, task: GoalplanTask): bool ); } +export function openDecisionIdsForPhase(plan: Goalplan, wp: GoalplanWorkPhase): string[] { + // Only a unique, decided target releases a reference. Missing/duplicate ids fail closed. + return [...new Set(wp.awaitsDecision ?? [])].filter((id) => { + const matches = (plan.decisions ?? []).filter((decision) => decision.id === id); + return !(matches.length === 1 && matches[0].status === "decided"); + }); +} + +function workPhaseReadyConditionsMet(plan: Goalplan, wp: GoalplanWorkPhase): boolean { + return workPhaseDependenciesMet(plan, wp) && openDecisionIdsForPhase(plan, wp).length === 0; +} + function isRunnablePhase(plan: Goalplan, wp: GoalplanWorkPhase): boolean { - return ( - (wp.status === "pending" || wp.status === "in_progress") - && workPhaseDependenciesMet(plan, wp) - ); + return (wp.status === "pending" || wp.status === "in_progress") + && workPhaseReadyConditionsMet(plan, wp); } export function readyWorkPhases(plan: Goalplan): GoalplanWorkPhase[] { @@ -1060,6 +1122,8 @@ export function dependencyWaitReasons(plan: Goalplan): string[] { unmetPhaseDependencies.map((id) => describePhaseDependency(plan, id)), )); } + const decisionIds = openDecisionIdsForPhase(plan, wp); + if (decisionIds.length > 0) reasons.push(`work-phase ${wp.id} awaits decision ${decisionIds.join(", ")}`); if (wp.status !== "pending" && wp.status !== "in_progress") continue; for (const task of wp.tasks.filter((candidate) => candidate.status === "pending")) { const unmetTaskDependencies = unmetTaskDependencyIds(wp, task); @@ -1097,6 +1161,8 @@ export function dependencyDeadlock(plan: Goalplan): DependencyDeadlock | null { reasons.push( `work-phase ${wp.id} is blocked${wp.blockedReason ? ` (${wp.blockedReason})` : ""}`, ); + const decisionIds = openDecisionIdsForPhase(plan, wp); + if (decisionIds.length > 0) reasons.push(`work-phase ${wp.id} awaits decision ${decisionIds.join(", ")}`); continue; } const unmetPhaseDependencies = unmetPhaseDependencyIds(plan, wp); @@ -1105,8 +1171,10 @@ export function dependencyDeadlock(plan: Goalplan): DependencyDeadlock | null { `work-phase ${wp.id}`, unmetPhaseDependencies.map((id) => describePhaseDependency(plan, id)), )); - continue; } + const decisionIds = openDecisionIdsForPhase(plan, wp); + if (decisionIds.length > 0) reasons.push(`work-phase ${wp.id} awaits decision ${decisionIds.join(", ")}`); + if (unmetPhaseDependencies.length > 0) continue; for (const task of wp.tasks.filter((candidate) => candidate.status === "pending")) { const unmetTaskDependencies = unmetTaskDependencyIds(wp, task); if (unmetTaskDependencies.length > 0) { @@ -1120,6 +1188,31 @@ export function dependencyDeadlock(plan: Goalplan): DependencyDeadlock | null { return reasons.length > 0 ? { reasons } : null; } +/** IDLE can yield only when actual open user decisions account for all remaining work. */ +export function remainingWorkAwaitsDecisions(plan: Goalplan): boolean { + const remaining = remainingWorkPhases(plan); + if (remaining.length === 0) return false; + const byId = new Map(plan.workPhases.map((phase) => [phase.id, phase])); + const waiting = (phase: GoalplanWorkPhase, visiting: Set): boolean => { + if (phase.status !== "pending" && phase.status !== "in_progress") return false; + if ((phase.awaitsDecision ?? []).some((id) => { + const matches = (plan.decisions ?? []).filter((decision) => decision.id === id); + return matches.length === 1 && matches[0].status === "open"; + })) return true; + if (visiting.has(phase.id)) return false; + visiting.add(phase.id); + const result = (phase.dependsOn ?? []).some((id) => { + const dependency = byId.get(id); + return dependency !== undefined && dependency.status !== "done" && waiting(dependency, visiting); + }); + visiting.delete(phase.id); + return result; + }; + const waitingPhases = remaining.filter((phase) => waiting(phase, new Set())); + if (waitingPhases.length !== remaining.length) return false; + return unmetCriteria(plan).every((criterion) => waitingPhases.some((phase) => phase.criteriaIds.includes(criterion.id))); +} + const LIFECYCLE_ID_RE = /^[a-z0-9][a-z0-9-]{0,39}$/; export type GoalplanLifecycleResult = @@ -1127,6 +1220,55 @@ export type GoalplanLifecycleResult = | { kind: "unchanged"; plan: Goalplan; reason: string } | { kind: "rejected"; reason: string }; +export function askGoalplanDecision( + plan: Goalplan, + input: { id: string; question: string; recommendation?: string; workPhaseIds: string[]; askedAt: string }, +): GoalplanLifecycleResult { + const id = input.id.trim(); + const question = input.question.trim(); + const recommendation = input.recommendation?.trim(); + const workPhaseIds = input.workPhaseIds.map((phaseId) => phaseId.trim()); + if (!LIFECYCLE_ID_RE.test(id)) return { kind: "rejected", reason: "decision id must be a short lowercase id, e.g. dec-1" }; + if (!question) return { kind: "rejected", reason: "decision question must not be empty" }; + if (input.recommendation !== undefined && !recommendation) return { kind: "rejected", reason: "decision recommendation must not be empty" }; + if (!validIsoTime(input.askedAt)) return { kind: "rejected", reason: "decision askedAt must be an ISO timestamp" }; + if (plan.decisions?.some((decision) => decision.id === id)) return { kind: "rejected", reason: `decision '${id}' is already in this plan` }; + const duplicate = plan.decisions?.find((decision) => decision.status === "open" && decision.question.trim() === question); + if (duplicate) return { kind: "rejected", reason: `question is already open as decision '${duplicate.id}'` }; + if (workPhaseIds.some((phaseId) => !phaseId) || new Set(workPhaseIds).size !== workPhaseIds.length) { + return { kind: "rejected", reason: "--work-phase requires distinct non-empty ids" }; + } + for (const phaseId of workPhaseIds) { + const phase = plan.workPhases.find((wp) => wp.id === phaseId); + if (!phase) return { kind: "rejected", reason: `work phase '${phaseId}' is not in this plan` }; + if (phase.status === "done" || phase.status === "superseded") { + return { kind: "rejected", reason: `work phase '${phaseId}' is ${phase.status} and cannot await a decision` }; + } + } + const decision: GoalplanDecision = { id, question, status: "open", askedAt: input.askedAt, + ...(recommendation === undefined ? {} : { recommendation }) }; + const next: Goalplan = { ...plan, decisions: [...(plan.decisions ?? []), decision], + workPhases: plan.workPhases.map((wp) => workPhaseIds.includes(wp.id) + ? { ...wp, awaitsDecision: [...(wp.awaitsDecision ?? []), id] } : wp) }; + const reasons = goalplanDefinitionIntegrityReasons(next); + return reasons.length ? { kind: "rejected", reason: reasons.join("; ") } : { kind: "changed", plan: next }; +} + +export function decideGoalplanDecision( + plan: Goalplan, id: string, answer: string, decidedAt: string, +): GoalplanLifecycleResult { + const decision = plan.decisions?.find((candidate) => candidate.id === id.trim()); + if (!decision) return { kind: "rejected", reason: `decision '${id.trim()}' is not in this plan` }; + if (!answer.trim()) return { kind: "rejected", reason: "decision answer must not be empty" }; + if (!validIsoTime(decidedAt)) return { kind: "rejected", reason: "decision decidedAt must be an ISO timestamp" }; + if (decision.status === "decided") return decision.answer === answer.trim() + ? { kind: "unchanged", plan, reason: `decision '${id.trim()}' is already decided` } + : { kind: "rejected", reason: `decision '${id.trim()}' already has a different answer` }; + const next: Goalplan = { ...plan, decisions: plan.decisions!.map((candidate) => candidate.id === decision.id + ? { ...candidate, status: "decided" as const, answer: answer.trim(), decidedAt } : candidate) }; + return { kind: "changed", plan: next }; +} + export function addGoalplanTask( plan: Goalplan, workPhaseId: string, @@ -1317,6 +1459,22 @@ export function goalplanDefinitionIntegrityReasons(plan: Goalplan): string[] { for (const id of duplicateIds(plan.workPhases.map((phase) => phase.id))) { reasons.push(`duplicate work phase id '${id}' makes dependency references ambiguous`); } + const decisionsById = new Map((plan.decisions ?? []).map((decision) => [decision.id, decision])); + for (const id of duplicateIds((plan.decisions ?? []).map((decision) => decision.id))) { + reasons.push(`duplicate decision id '${id}' makes awaitsDecision references ambiguous`); + } + for (const phase of plan.workPhases) { + for (const id of duplicateIds(phase.awaitsDecision ?? [])) { + reasons.push(`work phase ${phase.id} awaits decision '${id}' more than once`); + } + for (const id of new Set(phase.awaitsDecision ?? [])) { + const decision = decisionsById.get(id); + if (!decision) reasons.push(`work phase ${phase.id} awaits unknown decision '${id}'`); + else if (phase.status === "done" && decision.status === "open") { + reasons.push(`work phase ${phase.id} is done while decision ${id} is open`); + } + } + } for (const phase of plan.workPhases) { // 감사 라운드 1 BLOCKER 1: 같은 참조를 여러 번 쓴 dependsOn이 같은 사유를 반복하면 // goal-gate의 slice(0, 4)가 한 문장으로 네 칸을 채워 다른 진단을 가린다. wp2 reviver는 @@ -1795,8 +1953,11 @@ export function closeFixedWorkPhase( // status alone let a target through whose dependency turned blocked after the // marker was written — advanceWorkPhase() answers no_active there, and recovery // must not answer ok. - if (!workPhaseDependenciesMet(plan, current)) { - return { kind: "dependencies_unmet", unmet: unmetPhaseDependencyIds(plan, current) }; + if (!workPhaseReadyConditionsMet(plan, current)) { + return { kind: "dependencies_unmet", unmet: [ + ...unmetPhaseDependencyIds(plan, current), + ...openDecisionIdsForPhase(plan, current).map((id) => `decision:${id}`), + ] }; } // CYCLE-COMPLETION-01, unchanged wording and unchanged variant: an open task keeps @@ -1823,10 +1984,10 @@ export function closeFixedWorkPhase( let next: { id: string } | undefined; if (recordedNext === undefined) { const after = closedWorkPhases.slice(currentIdx + 1).find( - (wp) => wp.status === "pending" && workPhaseDependenciesMet(closedPlan, wp), + (wp) => wp.status === "pending" && isRunnablePhase(closedPlan, wp), ); next = after ?? closedWorkPhases.slice(0, currentIdx).find( - (wp) => wp.status === "pending" && workPhaseDependenciesMet(closedPlan, wp), + (wp) => wp.status === "pending" && isRunnablePhase(closedPlan, wp), ); } else if (recordedNext === null) { next = undefined; @@ -1871,11 +2032,11 @@ export function closeFixedWorkPhase( // and it can only do that against a normalized cursor. next = closedWorkPhases.find( (wp) => wp.id === plan.activeWorkPhaseId && wp.id !== workPhaseId - && wp.status === "in_progress" && workPhaseDependenciesMet(closedPlan, wp), + && wp.status === "in_progress" && workPhaseReadyConditionsMet(closedPlan, wp), ); } else if (named.status !== "pending" && named.status !== "in_progress") { return { kind: "successor_lost", successorId: recordedNext, reason: "not_runnable" }; - } else if (!workPhaseDependenciesMet(closedPlan, named)) { + } else if (!workPhaseReadyConditionsMet(closedPlan, named)) { return { kind: "successor_lost", successorId: recordedNext, reason: "dependencies_unmet" }; } else { next = named; @@ -1925,7 +2086,7 @@ export function absentSuccessorDetail( ? "is gone too" : reason === "not_runnable" ? "can no longer be started" - : "now waits for another work-phase"; + : "now waits for a prerequisite or decision"; } export type ResumeAbsentTargetResult = @@ -1954,7 +2115,7 @@ export function resumeAbsentTarget( // successor waiting on the same unmet dependency was refused with the target present and // activated with it gone. A dangling dependsOn reads as not-done here by design, and the // pending branch already refused that plan. - if (!workPhaseDependenciesMet(plan, named)) { + if (!workPhaseReadyConditionsMet(plan, named)) { return { kind: "successor_lost", successorId: recordedNext, reason: "dependencies_unmet" }; } // Running: the activation happened too, but only if the cursor agrees. §45 established @@ -2044,11 +2205,11 @@ export function effectiveActiveWorkPhaseId(plan: Goalplan): string | null { if (cur && isRunnablePhase(plan, cur)) return cur.id; } const inProgress = plan.workPhases.find( - (wp) => wp.status === "in_progress" && workPhaseDependenciesMet(plan, wp), + (wp) => wp.status === "in_progress" && isRunnablePhase(plan, wp), ); if (inProgress) return inProgress.id; const pending = plan.workPhases.find( - (wp) => wp.status === "pending" && workPhaseDependenciesMet(plan, wp), + (wp) => wp.status === "pending" && isRunnablePhase(plan, wp), ); return pending?.id ?? null; } diff --git a/plugins/codexclaw/components/pabcd-state/src/hook.ts b/plugins/codexclaw/components/pabcd-state/src/hook.ts index 85dcba0c..77b5cd2c 100644 --- a/plugins/codexclaw/components/pabcd-state/src/hook.ts +++ b/plugins/codexclaw/components/pabcd-state/src/hook.ts @@ -74,6 +74,7 @@ import { readGoalplan, readyTasks, readyWorkPhases, + remainingWorkAwaitsDecisions, unmetCriteria, withGoalplanWriteLock, writeGoalplan, @@ -1822,7 +1823,9 @@ export function handleStop( // plan gets a bounded arming block — "IDLE is not the end while work remains". if (!inFlight) { if (!goalActive) return ""; - if (!state.slug || !safeReadBoundGoalplan(payload.cwd, state.slug)) return ""; + const plan = state.slug ? safeReadBoundGoalplan(payload.cwd, state.slug) : null; + if (!plan) return ""; + if (remainingWorkAwaitsDecisions(plan)) return ""; // bail: don't pile on during context-pressure/compaction recovery. if (isContextPressureTail(readTranscriptTail(payload.transcript_path))) return ""; const count = bumpStopCounter(payload.cwd, state); diff --git a/plugins/codexclaw/components/pabcd-state/test/goalplan-public-surface.test.ts b/plugins/codexclaw/components/pabcd-state/test/goalplan-public-surface.test.ts index 26b20f11..f7741eb9 100644 --- a/plugins/codexclaw/components/pabcd-state/test/goalplan-public-surface.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/goalplan-public-surface.test.ts @@ -492,7 +492,7 @@ test("help lists repeated dependency syntax and required outcome", () => { // unknown-verb 거부 문구가 새 동사 넷을 포함하고 기존 여섯을 순서대로 남긴다. // 다음 verb 추가가 이 문구를 다시 빠뜨리면 여기서 RED가 난다. assert.deepEqual(parseGoalplanCliArgs(["redy"], "/tmp"), { - error: "unknown loop verb 'redy' (expected init|show|validate|steer|add-criterion|add-work-phase|ready|add-task|complete-task|meet-criterion); run cxc loop --help", + error: "unknown loop verb 'redy' (expected init|show|validate|steer|add-criterion|add-work-phase|ready|add-task|complete-task|meet-criterion|ask|decide); run cxc loop --help", }); }); @@ -680,3 +680,101 @@ test("the built CLI rejects a misplaced or misspelled flag before any write", () } } }); + +test("ask records an open decision and hides only linked work phases", () => { + const plan = fixture(); + plan.workPhases.push({ id: "wp-free", title: "free", status: "pending", tasks: [{ id: "free-task", title: "free", status: "pending" }], criteriaIds: [] }); + const { cwd, session } = workspace(plan); + const asked = cli(cwd, ["ask", "--session", session, "--id", "dec-1", "--question", "Choose API", "--recommendation", "Use v2", "--work-phase", "wp-live"]); + assert.equal(asked.code, 0, asked.output); + const back = readGoalplan(cwd, plan.slug)!; + assert.equal(back.decisions?.[0]?.status, "open"); + assert.match(back.decisions?.[0]?.askedAt ?? "", /^\d{4}-\d\d-\d\dT/); + assert.deepEqual(back.workPhases.find((wp) => wp.id === "wp-live")?.awaitsDecision, ["dec-1"]); + assert.deepEqual(readyWorkPhases(back).map((wp) => wp.id), ["wp-free"]); + assert.deepEqual(readyTasks(back).map(({ workPhaseId }) => workPhaseId), ["wp-free"]); + const ready = cli(cwd, ["ready", "--session", session, "--json"]); + assert.equal(ready.code, 0, ready.output); + const data = JSON.parse(ready.output); + assert.equal(data.openDecisions[0].id, "dec-1"); + assert.deepEqual(data.awaitingDecisions, [{ workPhaseId: "wp-live", decisionIds: ["dec-1"] }]); + assert.match(cli(cwd, ["show", "--session", session]).output, /Choose API[\s\S]*waiting: wp-live/); +}); + +test("decide releases linked phases without unblocking explicit blocks", () => { + const plan = fixture(); + plan.workPhases.push({ id: "wp-explicit", title: "explicit", status: "blocked", blockedReason: "vendor", tasks: [], criteriaIds: [] }); + const { cwd, session } = workspace(plan); + assert.equal(cli(cwd, ["ask", "--session", session, "--id", "dec-1", "--question", "Choose API", "--work-phase", "wp-live", "--work-phase", "wp-explicit"]).code, 0); + assert.equal(cli(cwd, ["decide", "--session", session, "--id", "dec-1", "--answer", "Use v2"]).code, 0); + const back = readGoalplan(cwd, plan.slug)!; + assert.equal(back.decisions?.[0]?.answer, "Use v2"); + assert.equal(back.decisions?.[0]?.status, "decided"); + assert.match(back.decisions?.[0]?.decidedAt ?? "", /^\d{4}-/); + assert.deepEqual(readyWorkPhases(back).map((wp) => wp.id), ["wp-live"]); + assert.equal(back.workPhases.find((wp) => wp.id === "wp-explicit")?.blockedReason, "vendor"); + assert.deepEqual(JSON.parse(cli(cwd, ["ready", "--session", session, "--json"]).output).awaitingDecisions, []); +}); + +test("ask rejects duplicate open question and unknown phase without a write", () => { + const plan = fixture(); + const { cwd, session } = workspace(plan); + assert.equal(cli(cwd, ["ask", "--session", session, "--id", "dec-1", "--question", "Choose API", "--work-phase", "wp-live"]).code, 0); + const before = planText(cwd, plan.slug), ledger = ledgerText(cwd, plan.slug); + const duplicate = cli(cwd, ["ask", "--session", session, "--id", "dec-2", "--question", " Choose API "]); + assert.equal(duplicate.code, 1); + assert.match(duplicate.output, /dec-1/); + const unknown = cli(cwd, ["ask", "--session", session, "--id", "dec-2", "--question", "Other", "--work-phase", "ghost"]); + assert.equal(unknown.code, 1); + assert.match(unknown.output, /ghost/); + assert.equal(planText(cwd, plan.slug), before); + assert.equal(ledgerText(cwd, plan.slug), ledger); +}); + +test("ask and decide enforce per-verb flags before writing", () => { + const plan = fixture(); + const { cwd, session } = workspace(plan); + const before = planText(cwd, plan.slug), ledger = ledgerText(cwd, plan.slug); + for (const argv of [ + ["ask", "--session", session, "--id", "dec-1", "--answer", "x"], + ["decide", "--session", session, "--id", "dec-1", "--question", "x"], + ["ask", "--session", session, "--id", "dec-1", "--question", "x", "--question", "y"], + ["ask", "--session", session, "--id", "dec-1", "--question", "x", "--work-phase", "wp-live", "--work-phase", "wp-live"], + ["ask", "--session", session, "--id", "dec-1", "--question="], + ["decide", "--session", session, "--id", "dec-1", "--answer"], + ]) assert.equal("error" in parseGoalplanCliArgs(argv, cwd), true, argv.join(" ")); + assert.equal(planText(cwd, plan.slug), before); + assert.equal(ledgerText(cwd, plan.slug), ledger); + assert.match(renderGoalplanHelp(), /ask --session --id --question /); + assert.match(renderGoalplanHelp(), /decide --session --id --answer /); +}); + +test("decide is idempotent only for the same answer", () => { + const plan = fixture(); + const { cwd, session } = workspace(plan); + cli(cwd, ["ask", "--session", session, "--id", "dec-1", "--question", "Choose API"]); + assert.equal(cli(cwd, ["decide", "--session", session, "--id", "dec-1", "--answer", "Use v2"]).code, 0); + const before = planText(cwd, plan.slug); + assert.equal(cli(cwd, ["decide", "--session", session, "--id", "dec-1", "--answer", "Use v2"]).code, 0); + assert.equal(planText(cwd, plan.slug), before); + assert.equal(cli(cwd, ["decide", "--session", session, "--id", "dec-1", "--answer", "Use v3"]).code, 1); + assert.equal(planText(cwd, plan.slug), before); +}); + +test("ready rejects dangling and duplicate decision references", () => { + const plan = fixture(); + const { cwd, session } = workspace(plan); + plan.workPhases.find((wp) => wp.id === "wp-live")!.awaitsDecision = ["ghost"]; + writeGoalplan(cwd, plan); + const dangling = cli(cwd, ["ready", "--session", session]); + assert.equal(dangling.code, 1); + assert.match(dangling.output, /awaits unknown decision 'ghost'/); + plan.decisions = [{ id: "ghost", question: "Choose", status: "open", askedAt: "2026-09-28T00:00:00.000Z" }]; + plan.workPhases.find((wp) => wp.id === "wp-live")!.awaitsDecision = ["ghost", "ghost"]; + writeGoalplan(cwd, plan); + assert.match(cli(cwd, ["ready", "--session", session]).output, /more than once/); + plan.workPhases.find((wp) => wp.id === "wp-live")!.awaitsDecision = ["ghost"]; + plan.decisions.push({ ...plan.decisions[0], question: "Again" }); + writeGoalplan(cwd, plan); + assert.match(cli(cwd, ["ready", "--session", session]).output, /duplicate decision id/); +}); diff --git a/plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts b/plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts index 45293de2..16664324 100644 --- a/plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts @@ -1451,3 +1451,85 @@ test("wp6: Stop reason keeps a single blocked phase when no work is ready", () = rmSync(cwd, { recursive: true, force: true }); } }); + +function decisionStopPlan(cwd: string, sessionId: string, phases: ReturnType["workPhases"], criteria: ReturnType["criteria"] = []): ReturnType { + const plan = buildGoalplan({ objective: `decision stop ${sessionId}` }); + plan.workPhases = phases; + plan.criteria = criteria; + plan.decisions = [{ id: "dec-1", question: "Choose API", status: "open", askedAt: "2026-09-28T00:00:00.000Z" }]; + writeGoalplan(cwd, plan); + writeState(cwd, { ...defaultState(sessionId), slug: plan.slug }); + return plan; +} + +const waitingPhase = (id: string, criteriaIds: string[] = []) => ({ id, title: id, status: "pending" as const, tasks: [], criteriaIds, awaitsDecision: ["dec-1"] }); + +test("IDLE Stop releases when every remaining phase awaits an open decision", () => { + const cwd = freshCwd(); + withGoalsDb([{ thread_id: "dec-idle", status: "active" }], () => { + decisionStopPlan(cwd, "dec-idle", [waitingPhase("linked")]); + assert.equal(handleStop(stop(cwd, "dec-idle")), ""); + assert.equal(readState(cwd, "dec-idle").stopBlockTotal, 0); + }); +}); + +test("IDLE Stop still blocks when an independent phase is runnable", () => { + const cwd = freshCwd(); + withGoalsDb([{ thread_id: "dec-free", status: "active" }], () => { + decisionStopPlan(cwd, "dec-free", [waitingPhase("linked"), { id: "free", title: "free", status: "pending", tasks: [], criteriaIds: [] }]); + assert.equal(JSON.parse(handleStop(stop(cwd, "dec-free")).trim()).decision, "block"); + }); +}); + +test("IDLE Stop blocks again after decide releases the wait", () => { + const cwd = freshCwd(); + withGoalsDb([{ thread_id: "dec-answered", status: "active" }], () => { + const plan = decisionStopPlan(cwd, "dec-answered", [waitingPhase("linked")]); + assert.equal(handleStop(stop(cwd, "dec-answered")), ""); + plan.decisions = [{ ...plan.decisions![0], status: "decided", answer: "Use v2", decidedAt: "2026-09-28T01:00:00.000Z" }]; + writeGoalplan(cwd, plan); + assert.match(JSON.parse(handleStop(stop(cwd, "dec-answered")).trim()).reason, /cxc orchestrate P/); + }); +}); + +test("IDLE Stop still blocks when one phase waits on a decision and another is blocked for another reason", () => { + const cwd = freshCwd(); + withGoalsDb([{ thread_id: "dec-blocked", status: "active" }], () => { + decisionStopPlan(cwd, "dec-blocked", [waitingPhase("linked"), { id: "blocked", title: "blocked", status: "blocked", blockedReason: "vendor", tasks: [], criteriaIds: [] }]); + assert.equal(JSON.parse(handleStop(stop(cwd, "dec-blocked")).trim()).decision, "block"); + }); +}); + +test("IDLE Stop releases when an in-progress phase gained an open decision mid-cycle and its dependents wait on it", () => { + const cwd = freshCwd(); + withGoalsDb([{ thread_id: "dec-chain", status: "active" }], () => { + decisionStopPlan(cwd, "dec-chain", [{ ...waitingPhase("root"), status: "in_progress" }, { id: "child", title: "child", status: "pending", tasks: [], criteriaIds: [], dependsOn: ["root"] }]); + assert.equal(handleStop(stop(cwd, "dec-chain")), ""); + }); +}); + +test("IDLE Stop still blocks when an independent criterion is unmet", () => { + const cwd = freshCwd(); + withGoalsDb([{ thread_id: "dec-criterion", status: "active" }], () => { + decisionStopPlan(cwd, "dec-criterion", [waitingPhase("linked")], [{ id: "c-1", scenario: "independent", expectedEvidence: "proof", capturedEvidence: null, status: "open" }]); + assert.equal(JSON.parse(handleStop(stop(cwd, "dec-criterion")).trim()).decision, "block"); + }); +}); + +test("IDLE Stop releases when every unmet criterion belongs to a decision-waiting phase", () => { + const cwd = freshCwd(); + withGoalsDb([{ thread_id: "dec-criterion-linked", status: "active" }], () => { + decisionStopPlan(cwd, "dec-criterion-linked", [waitingPhase("linked", ["c-1"])], [{ id: "c-1", scenario: "linked", expectedEvidence: "proof", capturedEvidence: null, status: "open" }]); + assert.equal(handleStop(stop(cwd, "dec-criterion-linked")), ""); + }); +}); + +test("dangling decision reference does not release IDLE Stop", () => { + const cwd = freshCwd(); + withGoalsDb([{ thread_id: "dec-dangling", status: "active" }], () => { + const plan = decisionStopPlan(cwd, "dec-dangling", [waitingPhase("linked")]); + plan.decisions = []; + writeGoalplan(cwd, plan); + assert.equal(JSON.parse(handleStop(stop(cwd, "dec-dangling")).trim()).decision, "block"); + }); +}); diff --git a/plugins/codexclaw/components/pabcd-state/test/orchestrate-cli.test.ts b/plugins/codexclaw/components/pabcd-state/test/orchestrate-cli.test.ts index adcb0a74..955a3480 100644 --- a/plugins/codexclaw/components/pabcd-state/test/orchestrate-cli.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/orchestrate-cli.test.ts @@ -1866,7 +1866,7 @@ test("an absent target refuses a running successor whose dependency is unmet", ( const result = runOrchestrateCli(parsedDclose(cwd, id)); assert.equal(result.code, 1, result.output); - assert.match(result.output, /now waits for another work-phase/); + assert.match(result.output, /now waits for a prerequisite or decision/); // Fail closed: no plan write, no ledger row, and the marker stays for a real repair. assert.equal(readFileSync(join(cwd, ".codexclaw/goalplans", slug, "goalplan.json"), "utf8"), before); assert.equal(readState(cwd, id).dcloseRecovery?.nextWorkPhaseId, "wp-2"); diff --git a/plugins/codexclaw/components/pabcd-state/test/work-phase-states.test.ts b/plugins/codexclaw/components/pabcd-state/test/work-phase-states.test.ts index 1631a73c..babbe36a 100644 --- a/plugins/codexclaw/components/pabcd-state/test/work-phase-states.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/work-phase-states.test.ts @@ -14,6 +14,11 @@ import { join } from "node:path"; import { advanceWorkPhase, buildGoalplan, + closeFixedWorkPhase, + resumeAbsentTarget, + goalplanDefinitionIntegrityReasons, + readGoalplanDetailed, + remainingWorkAwaitsDecisions, dependencyDeadlock, dependencyWaitReasons, effectiveActiveWorkPhaseId, @@ -21,6 +26,7 @@ import { nextOpenTask, readGoalplan, remainingWorkPhases, + readyWorkPhases, validateGoalplan, writeGoalplan, type Goalplan, @@ -324,3 +330,118 @@ test("wp4: dependency wait reasons include phase and task waits", () => { "task build/t-dependent waits for task build/t-upstream (pending)", ]); }); + +test("open decision excludes linked phase from cursor close and successor selection", () => { + const p = plan([ + phase("linked", "in_progress", { awaitsDecision: ["dec-1"] }), + phase("free", "pending"), + ], { activeWorkPhaseId: "linked", decisions: [{ id: "dec-1", question: "Choose", status: "open", askedAt: "2026-09-28T00:00:00.000Z" }] }); + assert.equal(effectiveActiveWorkPhaseId(p), "free"); + const advanced = advanceWorkPhase(p); + assert.equal(advanced.kind, "ok"); + if (advanced.kind === "ok") { + assert.equal(advanced.plan.workPhases[0].status, "in_progress"); + assert.equal(advanced.plan.workPhases[1].status, "done"); + assert.match(dependencyDeadlock(advanced.plan)?.reasons.join(" ") ?? "", /linked awaits decision dec-1/); + } +}); + +test("decision wait reasons appear beside ready independent work", () => { + const p = plan([phase("free", "pending"), phase("linked", "pending", { awaitsDecision: ["dec-1"] })], + { decisions: [{ id: "dec-1", question: "Choose", status: "open", askedAt: "2026-09-28T00:00:00.000Z" }] }); + assert.equal(dependencyDeadlock(p), null); + assert.match(dependencyWaitReasons(p).join(" "), /work-phase linked awaits decision dec-1/); +}); + +test("legacy plan round trips without decision fields", () => { + const p = plan([phase("free", "pending")]); + const back = roundTrip(p)!; + assert.equal("decisions" in back, false); + assert.equal("awaitsDecision" in back.workPhases[0], false); + assert.deepEqual(readyWorkPhases(back).map((wp) => wp.id), ["free"]); +}); + +test("E8 fails a done phase waiting on an open decision but permits an unrelated open decision", () => { + const d = { id: "dec-1", question: "Choose", status: "open" as const, askedAt: "2026-09-28T00:00:00.000Z" }; + const linked = plan([phase("done", "done", { awaitsDecision: ["dec-1"] })], { decisions: [d] }); + assert.match(validateGoalplan(linked).reasons.join(" "), /done while decision dec-1 is open/); + const unlinked = plan([phase("done", "done")], { decisions: [d] }); + assert.equal(validateGoalplan(unlinked).ok, true); + const pending = plan([phase("linked", "pending", { awaitsDecision: ["dec-1"] })], { decisions: [d] }); + assert.match(validateGoalplan(pending).reasons.join(" "), /work phase\(s\) not done/); +}); + +test("missing decision id keeps the phase waiting", () => { + const p = plan([phase("linked", "pending", { awaitsDecision: ["ghost"] })], { activeWorkPhaseId: "linked" }); + assert.equal(effectiveActiveWorkPhaseId(p), null); +}); + +test("duplicate decision id keeps the phase waiting", () => { + const decided = { id: "dec-1", question: "Choose", status: "decided" as const, answer: "yes", askedAt: "2026-09-28T00:00:00.000Z", decidedAt: "2026-09-28T01:00:00.000Z" }; + const open = { id: "dec-1", question: "Again", status: "open" as const, askedAt: "2026-09-28T00:00:00.000Z" }; + for (const decisions of [[decided, open], [open, decided]]) { + const p = plan([phase("linked", "pending", { awaitsDecision: ["dec-1"] })], { activeWorkPhaseId: "linked", decisions }); + assert.equal(effectiveActiveWorkPhaseId(p), null); + } +}); + +test("invalid decision fields and dangling references fail closed", () => { + const cwd = mkdtempSync(join(tmpdir(), "cxc-dec-invalid-")); + const p = plan([phase("linked", "pending", { awaitsDecision: ["dec-1"] })], + { decisions: [{ id: "dec-1", question: "Choose", status: "open", askedAt: "2026-09-28T00:00:00.000Z" }] }); + writeGoalplan(cwd, p); + const file = join(goalplanDir(cwd, p.slug), "goalplan.json"); + const original = JSON.parse(readFileSync(file, "utf8")) as Goalplan; + for (const [change, field] of [ + [(raw: Goalplan) => { raw.decisions = [{ ...raw.decisions![0], askedAt: "bad" }]; }, "decisions"], + [(raw: Goalplan) => { raw.workPhases[0].awaitsDecision = [""]; }, "workPhases[].awaitsDecision"], + ] as const) { + const raw = structuredClone(original); + change(raw); + writeFileSync(file, JSON.stringify(raw)); + assert.equal(readGoalplan(cwd, p.slug), null); + assert.equal((readGoalplanDetailed(cwd, p.slug).diagnostic as { field: string }).field, field); + } + writeFileSync(file, JSON.stringify(original)); + const dangling = { ...p, decisions: [] }; + assert.match(goalplanDefinitionIntegrityReasons(dangling).join(" "), /awaits unknown decision 'dec-1'/); + assert.deepEqual(readyWorkPhases(dangling), []); + assert.match(goalplanDefinitionIntegrityReasons({ ...p, decisions: [p.decisions![0], p.decisions![0]] }).join(" "), /duplicate decision id/); + assert.match(goalplanDefinitionIntegrityReasons({ ...p, workPhases: [phase("linked", "pending", { awaitsDecision: ["dec-1", "dec-1"] })] }).join(" "), /more than once/); +}); + +test("decision-waiting successor cannot activate through close or absent-target recovery", () => { + const d = { id: "dec-1", question: "Choose", status: "open" as const, askedAt: "2026-09-28T00:00:00.000Z" }; + const p = plan([phase("current", "in_progress"), phase("linked", "pending", { awaitsDecision: ["dec-1"] })], + { activeWorkPhaseId: "current", decisions: [d] }); + const closed = closeFixedWorkPhase(p, "current"); + assert.equal(closed.kind, "ok"); + if (closed.kind === "ok") assert.equal(closed.plan.activeWorkPhaseId, null); + assert.equal(closeFixedWorkPhase(p, "current", "linked").kind, "successor_lost"); + assert.equal(resumeAbsentTarget(p, "linked").kind, "successor_lost"); + const currentWaits = { ...p, workPhases: [phase("current", "in_progress", { awaitsDecision: ["dec-1"] }), p.workPhases[1]] }; + assert.equal(closeFixedWorkPhase(currentWaits, "current").kind, "dependencies_unmet"); +}); + +test("duplicate decision id blocks successor and absent-target recovery in either order", () => { + const decided = { id: "dec-1", question: "Choose", status: "decided" as const, answer: "yes", askedAt: "2026-09-28T00:00:00.000Z", decidedAt: "2026-09-28T01:00:00.000Z" }; + const open = { id: "dec-1", question: "Again", status: "open" as const, askedAt: "2026-09-28T00:00:00.000Z" }; + for (const decisions of [[decided, open], [open, decided]]) { + const p = plan([phase("current", "in_progress"), phase("linked", "pending", { awaitsDecision: ["dec-1"] })], { decisions }); + assert.equal(closeFixedWorkPhase(p, "current").kind, "ok"); + assert.equal(closeFixedWorkPhase(p, "current", "linked").kind, "successor_lost"); + assert.equal(resumeAbsentTarget(p, "linked").kind, "successor_lost"); + } +}); + +test("remaining work awaits decisions only for actual open answers and covered criteria", () => { + const d = { id: "dec-1", question: "Choose", status: "open" as const, askedAt: "2026-09-28T00:00:00.000Z" }; + const root = phase("root", "in_progress", { awaitsDecision: ["dec-1"], criteriaIds: ["c-1"] }); + const child = phase("child", "pending", { dependsOn: ["root"] }); + const criterion = { id: "c-1", scenario: "chosen", expectedEvidence: "proof", capturedEvidence: null, status: "open" as const }; + const p = plan([root, child], { decisions: [d], criteria: [criterion] }); + assert.equal(remainingWorkAwaitsDecisions(p), true); + assert.equal(remainingWorkAwaitsDecisions({ ...p, decisions: [] }), false); + assert.equal(remainingWorkAwaitsDecisions({ ...p, criteria: [...p.criteria, { ...criterion, id: "c-2" }] }), false); + assert.equal(remainingWorkAwaitsDecisions({ ...p, workPhases: [...p.workPhases, phase("free", "pending")] }), false); +}); From 1673ce56b8b62fb8bd5708a5cba9bbcda838a2a5 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 03:15:03 +0900 Subject: [PATCH 58/90] Document goalplan question and answer workflow --- plugins/codexclaw/skills/dev/references/async-questions.md | 2 ++ .../codexclaw/skills/loop/references/durable-goalplan.md | 7 ++++++- 2 files changed, 8 insertions(+), 1 deletion(-) diff --git a/plugins/codexclaw/skills/dev/references/async-questions.md b/plugins/codexclaw/skills/dev/references/async-questions.md index 0e9dfb1a..0d38e95b 100644 --- a/plugins/codexclaw/skills/dev/references/async-questions.md +++ b/plugins/codexclaw/skills/dev/references/async-questions.md @@ -53,6 +53,8 @@ earlier answer must wait. Do not inherit the blocking tool's three-question limi within the authorized scope. A missing reply never blocks completion of optional work. Required input or approval remains a real dependency: silence and preselection are never consent; only the dependent action stays pending. +For a bound goalplan, after the host confirms a question was submitted, run `cxc loop ask --session --id --question [--recommendation ] [--work-phase ]...` to keep it in the existing plan. Record each dependent work phase. Check `cxc loop show` before sending a similar question after compaction. When the user answers, run `cxc loop decide --session --id --answer ` and resume only newly runnable phases. `ask` does not deliver a question; CLI success is no proof of host submission. Link a phase only while its action truly requires the reply. + 4. A later user message supplies the answer. Match it to the pending decision, update assumptions and affected work, and retain the original objective unless the user changes it. An ambiguous reply leaves the unresolved part pending. diff --git a/plugins/codexclaw/skills/loop/references/durable-goalplan.md b/plugins/codexclaw/skills/loop/references/durable-goalplan.md index f6a490bf..de80b491 100644 --- a/plugins/codexclaw/skills/loop/references/durable-goalplan.md +++ b/plugins/codexclaw/skills/loop/references/durable-goalplan.md @@ -49,7 +49,7 @@ This is the on-disk shape under `.codexclaw/goalplans//goalplan.json` - `objective`, `slug`, `createdAt`, `updatedAt`. - `workPhases[]` — each `{ id, title, status: pending|in_progress|done, dependsOn?, tasks[], criteriaIds[] }`. - `workPhase.dependsOn` names prerequisite work phases. `activeWorkPhaseId` marks the current one. + `workPhase.dependsOn` names prerequisite work phases. Optional `workPhase.awaitsDecision?: string[]` names decisions that must be answered before this phase can run. `activeWorkPhaseId` marks the current one. `workPhases[]` is APPEND-friendly mid-loop: when a new independent unit is discovered (LOOP-UNIT-CHAIN-01), add its work-phase (+ criteria) as a P-phase amendment instead of treating the plan as frozen at init or ending the goal. @@ -57,6 +57,7 @@ This is the on-disk shape under `.codexclaw/goalplans//goalplan.json` Task ids and task dependency references are phase-local: `task.dependsOn` names existing task ids in the same work phase, never a task in another phase. A done task carries a non-empty `outcome`; a pending task has no outcome. +- Optional `decisions[]` — each `{ id, question, recommendation?, status: open|decided, answer?, askedAt, decidedAt? }`. Open decisions have no answer or decidedAt; decided decisions require both. Decision ids are short lowercase ids. Absent and empty arrays remain distinct on disk, as do absent and empty `awaitsDecision` arrays. Old plans acquire neither field on read/write. Only linked pending or in-progress phases wait; an unrelated open decision does not pause the goal. - `criteria[]` — each `{ id, scenario, surface, presented?, expectedEvidence, capturedEvidence, status: open|met }`. `scenario` is the `--criterion` text and `surface` is one of `logic` (default), `web`, `tui` or `desktop`, set by `add-criterion --surface` on a session-bound plan @@ -94,12 +95,16 @@ This is the on-disk shape under `.codexclaw/goalplans//goalplan.json` - `cxc loop ready (--slug | --objective | --session ) [--json] [--cwd ]` - `cxc loop add-task --session --work-phase --id --title [--depends-on ]... [--cwd ]` - `cxc loop complete-task --session --work-phase --id --outcome [--cwd ]` +- `cxc loop ask --session --id --question [--recommendation ] [--work-phase ]... [--cwd ]` — record a question after sending it through the host. It never sends a message. Name each dependent phase. +- `cxc loop decide --session --id --answer [--cwd ]` — record the user reply. This changes only the decision record; phase status and blockedReason stay as they were. - `cxc loop meet-criterion --session --id --evidence [--cwd ]` — `--id` takes a generated `c-N` id; read it from `cxc loop show` or the goalplan file. - `cxc goalplan *` — deprecated alias for the same behavior during migration. The parser rejects unknown flags, stray positionals, missing values, and flags belonging to another verb before dispatch. Every value flag also accepts `--flag=value`, which is the way to pass a value that starts with `--`. +`ready --json` includes `openDecisions` and `awaitingDecisions` when the plan has a decisions field; `show` displays each open question and its waiting phases. At IDLE, Stop releases when every remaining phase and unmet criterion waits on an open user decision; the goal remains active and completion is still gated. + Repeat `--depends-on` once per prerequisite; comma-separated values are one id. Existing dependencies are not edited after creation. `complete-task` and `meet-criterion` require non-empty proof text. From 5b65c18a25d441c196974b39cae5e3acc5f629cc Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 03:15:44 +0900 Subject: [PATCH 59/90] build: regenerate dist for goalplan decisions --- .../pabcd-state/dist/goalplan-cli.js | 92 ++++++++- .../components/pabcd-state/dist/goalplan.js | 191 ++++++++++++++++-- .../components/pabcd-state/dist/hook.js | 5 +- 3 files changed, 268 insertions(+), 20 deletions(-) diff --git a/plugins/codexclaw/components/pabcd-state/dist/goalplan-cli.js b/plugins/codexclaw/components/pabcd-state/dist/goalplan-cli.js index 80698288..17aefe6f 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/goalplan-cli.js +++ b/plugins/codexclaw/components/pabcd-state/dist/goalplan-cli.js @@ -13,6 +13,9 @@ */ import { addGoalplanTask, + askGoalplanDecision, + decideGoalplanDecision, + openDecisionIdsForPhase, buildGoalplan, completeGoalplanTask, goalplanDefinitionIntegrityReasons, @@ -106,6 +109,12 @@ import { applySteeringBatch } from "./steering.js"; + + + + + + @@ -124,6 +133,8 @@ const VERBS = new Set ([ "add-task", "complete-task", "meet-criterion", + "ask", + "decide", ]); @@ -137,6 +148,7 @@ const VERBS = new Set ([ + const VERB_RULES = { init: { allowed: new Set(["--objective", "--session", "--criterion", "--schema-version", "--cwd"]), repeatable: new Set(["--criterion"]), usage: "init --objective [--session ] [--criterion ]... [--schema-version ] [--cwd ]" }, show: { allowed: new Set(["--slug", "--objective", "--session", "--cwd"]), repeatable: new Set(), usage: "show (--slug | --objective | --session ) [--cwd ]" }, @@ -148,6 +160,8 @@ const VERB_RULES = { "add-task": { allowed: new Set(["--session", "--work-phase", "--id", "--title", "--depends-on", "--cwd"]), repeatable: new Set(["--depends-on"]), usage: "add-task --session --work-phase --id --title [--depends-on ]... [--cwd ]" }, "complete-task": { allowed: new Set(["--session", "--work-phase", "--id", "--outcome", "--cwd"]), repeatable: new Set(), usage: "complete-task --session --work-phase --id --outcome [--cwd ]" }, "meet-criterion": { allowed: new Set(["--session", "--id", "--evidence", "--cwd"]), repeatable: new Set(), usage: "meet-criterion --session --id --evidence [--cwd ]" }, + ask: { allowed: new Set(["--session", "--id", "--question", "--recommendation", "--work-phase", "--cwd"]), repeatable: new Set(["--work-phase"]), usage: "ask --session --id --question [--recommendation ] [--work-phase ]... [--cwd ]" }, + decide: { allowed: new Set(["--session", "--id", "--answer", "--cwd"]), repeatable: new Set(), usage: "decide --session --id --answer [--cwd ]" }, help: { allowed: new Set(), repeatable: new Set(), usage: "--help" }, }; @@ -163,12 +177,12 @@ export function parseGoalplanCliArgs(argv , cwd ) } if (!VERBS.has(verb)) { return { - error: `unknown loop verb '${argv[0] ?? ""}' (expected init|show|validate|steer|add-criterion|add-work-phase|ready|add-task|complete-task|meet-criterion); run cxc loop --help`, + error: `unknown loop verb '${argv[0] ?? ""}' (expected init|show|validate|steer|add-criterion|add-work-phase|ready|add-task|complete-task|meet-criterion|ask|decide); run cxc loop --help`, }; } const selected = verb ; const rule = VERB_RULES[selected]; - const out = { verb: selected, cwd, criteria: [], dependsOn: [] }; + const out = { verb: selected, cwd, criteria: [], dependsOn: [], workPhaseIds: [] }; const seen = new Set (); const reject = (message ) => ({ error: `${selected}: ${message}` }); for (let i = 1; i < argv.length; i++) { @@ -213,7 +227,17 @@ export function parseGoalplanCliArgs(argv , cwd ) case "--presented": out.presented = value; break; case "--id": out.id = value; break; case "--title": out.title = value; break; - case "--work-phase": out.workPhaseId = value; break; + case "--work-phase": { + if (selected !== "ask") { out.workPhaseId = value; break; } + const phaseId = value.trim(); + if (!phaseId) return reject("--work-phase requires one non-empty id"); + if (out.workPhaseIds .includes(phaseId)) return reject(`--work-phase must not repeat id '${phaseId}'`); + out.workPhaseIds .push(phaseId); + break; + } + case "--question": out.question = value; break; + case "--recommendation": out.recommendation = value; break; + case "--answer": out.answer = value; break; case "--outcome": out.outcome = value; break; case "--schema-version": { const parsed = Number(value); @@ -438,6 +462,13 @@ function runReady(args , plan ) { const phases = readyWorkPhases(plan); const tasks = readyTasks(plan); + const openDecisions = (plan.decisions ?? []).filter((decision) => decision.status === "open") + .map(({ id, question, recommendation, askedAt }) => ({ id, question, ...(recommendation === undefined ? {} : { recommendation }), askedAt })); + const awaitingDecisions = plan.workPhases + .filter((wp) => wp.status === "pending" || wp.status === "in_progress") + .map((wp) => ({ workPhaseId: wp.id, decisionIds: openDecisionIdsForPhase(plan, wp).filter((id) => + (plan.decisions ?? []).some((decision) => decision.id === id && decision.status === "open")) })) + .filter((entry) => entry.decisionIds.length > 0); if (args.json === true) { return { output: JSON.stringify({ @@ -455,6 +486,7 @@ function runReady(args , plan ) { id: entry.task.id, title: entry.task.title, })), + ...(plan.decisions === undefined ? {} : { openDecisions, awaitingDecisions }), }), code: 0, }; @@ -467,9 +499,52 @@ function runReady(args , plan ) { lines.push(tasks.length > 0 ? `readyTasks: ${tasks.map((entry) => `${entry.workPhaseId}/${entry.task.id} (${entry.task.title})`).join("; ")}` : "readyTasks: none"); + if (plan.decisions !== undefined) { + lines.push(openDecisions.length > 0 + ? `openDecisions: ${openDecisions.map((decision) => `${decision.id} (${decision.question})`).join("; ")}` + : "openDecisions: none"); + lines.push(awaitingDecisions.length > 0 + ? `awaitingDecisions: ${awaitingDecisions.map((entry) => `${entry.workPhaseId}: ${entry.decisionIds.join(", ")}`).join("; ")}` + : "awaitingDecisions: none"); + } return { output: lines.join("\n"), code: 0 }; } +/** Record a host-submitted question or its answer under the goalplan write lock. */ +function runDecision(args ) { + const session = (args.session ?? "").trim(); + if (!session) return { output: `loop ${args.verb}: --session is required`, code: 1 }; + if (!isCanonicalSessionId(session)) return { output: `loop ${args.verb}: session id is not canonical`, code: 1 }; + const slug = readState(args.cwd, session).slug; + if (!slug) return { output: `loop ${args.verb}: session '${session}' has no bound goalplan - run \`cxc loop init --session ${session}\` first`, code: 1 }; + const id = (args.id ?? "").trim(); + if (args.verb === "ask" && (!id || !(args.question ?? "").trim())) { + return { output: "loop ask: --id and non-empty --question are required", code: 1 }; + } + if (args.verb === "decide" && (!id || !(args.answer ?? "").trim())) { + return { output: "loop decide: --id and non-empty --answer are required", code: 1 }; + } + + const locked = withGoalplanWriteLock (args.cwd, slug, (plan) => { + const result = args.verb === "ask" + ? askGoalplanDecision(plan, { + id, question: args.question , recommendation: args.recommendation, + workPhaseIds: args.workPhaseIds ?? [], askedAt: new Date().toISOString(), + }) + : decideGoalplanDecision(plan, id, args.answer , new Date().toISOString()); + if (result.kind === "rejected") return { kind: "rejected", reason: result.reason }; + if (result.kind === "unchanged") return { kind: "unchanged", reason: result.reason }; + writeGoalplan(args.cwd, result.plan); + return { kind: "changed" }; + }); + if (locked.kind === "locked" || locked.kind === "unreadable") { + return { output: `loop ${args.verb}: ${locked.reason}`, code: 1 }; + } + if (locked.value.kind === "rejected") return { output: `loop ${args.verb}: ${locked.value.reason}`, code: 1 }; + if (locked.value.kind === "unchanged") return { output: `loop ${args.verb}: ${locked.value.reason}; nothing to do`, code: 0 }; + return { output: `loop ${args.verb}: ${slug} ${id} applied`, code: 0 }; +} + /** * 060 wp6: the three lifecycle verbs share one locked read-modify-write. * @@ -610,6 +685,12 @@ function renderPlanLines(plan , lock ) for (const c of plan.criteria) { lines.push(` - ${c.id} [${c.status}] ${c.scenario}`); } + for (const decision of plan.decisions ?? []) { + if (decision.status !== "open") continue; + lines.push(` - ${decision.id} [open] ${decision.question}`); + const waiting = plan.workPhases.filter((wp) => wp.awaitsDecision?.includes(decision.id)); + lines.push(` waiting: ${waiting.map((wp) => wp.id).join(", ") || "none"}`); + } return lines.join("\n"); } @@ -623,7 +704,7 @@ export function renderGoalplanHelp() { "cxc loop — durable goalplan for a multi-cycle PABCD loop", "", "Usage:", - ...(["init", "show", "validate", "steer", "add-criterion", "add-work-phase", "ready", "add-task", "complete-task", "meet-criterion", "help"] ) + ...(["init", "show", "validate", "steer", "add-criterion", "add-work-phase", "ready", "add-task", "complete-task", "meet-criterion", "ask", "decide", "help"] ) .map((verb) => ` cxc loop ${VERB_RULES[verb].usage}`), "", "Notes:", @@ -639,6 +720,8 @@ export function renderGoalplanHelp() { " additionally require an approved finalGate, and no verb in this build opens a", " final-gate review round, so opt in only if you can record that gate yourself.", " meet-criterion requires non-empty captured evidence for the same reason.", + " Send the question through the host first, then record it with ask; ask never sends a message.", + " Record the user's reply with decide. It changes only the decision record.", "", "steer --batch-json expects an object with:", ' { "idempotencyKey": "", "rationale": "", "evidence": "",', @@ -710,6 +793,7 @@ export function runGoalplanCli(args ) { } if (args.verb === "steer") return runSteer(args); + if (args.verb === "ask" || args.verb === "decide") return runDecision(args); if (args.verb === "add-criterion" || args.verb === "add-work-phase") return runAddOp(args); diff --git a/plugins/codexclaw/components/pabcd-state/dist/goalplan.js b/plugins/codexclaw/components/pabcd-state/dist/goalplan.js index 8d211227..a20ac94b 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/goalplan.js +++ b/plugins/codexclaw/components/pabcd-state/dist/goalplan.js @@ -138,6 +138,18 @@ export const DEFAULT_NEW_SCHEMA_VERSION = 1; + + + + + + + + + + + + @@ -272,6 +284,7 @@ export const DEFAULT_NEW_SCHEMA_VERSION = 1; + const MAX_SLUG_BYTES = 128; @@ -504,6 +517,37 @@ function reviveDependsOn(value ) { return ids; } +function validIsoTime(value ) { + if (typeof value !== "string") return false; + const date = new Date(value); + return Number.isFinite(date.valueOf()) && date.toISOString() === value; +} + +function reviveDecisions(value ) { + if (value === undefined) return undefined; + if (!Array.isArray(value)) return "invalid"; + const decisions = []; + for (const item of value) { + if (typeof item !== "object" || item === null || Array.isArray(item)) return "invalid"; + const d = item ; + if (typeof d.id !== "string" || !LIFECYCLE_ID_RE.test(d.id) + || typeof d.question !== "string" || !d.question.trim() + || !validIsoTime(d.askedAt) + || (d.recommendation !== undefined && (typeof d.recommendation !== "string" || !d.recommendation.trim()))) return "invalid"; + if (d.status === "open") { + if (d.answer !== undefined || d.decidedAt !== undefined) return "invalid"; + decisions.push({ id: d.id, question: d.question, status: "open", askedAt: d.askedAt, + ...(d.recommendation === undefined ? {} : { recommendation: d.recommendation }) }); + } else if (d.status === "decided") { + if (typeof d.answer !== "string" || !d.answer.trim() || !validIsoTime(d.decidedAt)) return "invalid"; + decisions.push({ id: d.id, question: d.question, status: "decided", answer: d.answer, + askedAt: d.askedAt, decidedAt: d.decidedAt, + ...(d.recommendation === undefined ? {} : { recommendation: d.recommendation }) }); + } else return "invalid"; + } + return decisions; +} + /** Best-effort structural validation; a malformed object reads as absent (null). */ function reviveGoalplan(parsed , expectedSlug ) { if (typeof parsed !== "object" || parsed === null || Array.isArray(parsed)) return null; @@ -525,6 +569,8 @@ function reviveGoalplan(parsed , expectedSlug ) if (typeof w.id !== "string" || typeof w.title !== "string") return null; const phaseDependsOn = reviveDependsOn(w.dependsOn); if (phaseDependsOn === "invalid") return null; + const awaitsDecision = reviveDependsOn(w.awaitsDecision); + if (awaitsDecision === "invalid") return null; const status = w.status === "in_progress" || w.status === "done" || w.status === "blocked" || w.status === "superseded" ? w.status @@ -548,6 +594,7 @@ function reviveGoalplan(parsed , expectedSlug ) : []; const phase = { id: w.id, title: w.title, status, tasks, criteriaIds }; if (phaseDependsOn !== undefined) phase.dependsOn = phaseDependsOn; + if (awaitsDecision !== undefined) phase.awaitsDecision = awaitsDecision; if (typeof w.blockedReason === "string") phase.blockedReason = w.blockedReason; if (typeof w.supersededBy === "string") phase.supersededBy = w.supersededBy; workPhases.push(phase); @@ -580,6 +627,8 @@ function reviveGoalplan(parsed , expectedSlug ) }; const reviewRounds = reviveReviewRounds(o.reviewRounds); + const decisions = reviveDecisions(o.decisions); + if (decisions === "invalid") return null; const plan = { objective: o.objective, @@ -594,6 +643,7 @@ function reviveGoalplan(parsed , expectedSlug ) // Only attach the 010 fields when they are actually present, so a plan written // before this feature round-trips byte-identical. if (reviewRounds !== undefined) plan.reviewRounds = reviewRounds; + if (decisions !== undefined) plan.decisions = decisions; if (typeof o.activePlanAuditRoundId === "string") plan.activePlanAuditRoundId = o.activePlanAuditRoundId; if (typeof o.activeFinalGateRoundId === "string") plan.activeFinalGateRoundId = o.activeFinalGateRoundId; if (typeof o.schemaVersion === "number" && Number.isFinite(o.schemaVersion)) { @@ -822,6 +872,7 @@ function firstInvalidField(parsed ) { for (const rawWp of o.workPhases) { const wp = rawWp ; if (reviveDependsOn(wp.dependsOn) === "invalid") return "workPhases[].dependsOn"; + if (reviveDependsOn(wp.awaitsDecision) === "invalid") return "workPhases[].awaitsDecision"; for (const rawTask of Array.isArray(wp.tasks) ? wp.tasks : []) { if (typeof rawTask !== "object" || rawTask === null) continue; const task = rawTask ; @@ -839,6 +890,7 @@ function firstInvalidField(parsed ) { if (typeof o.host !== "object" || o.host === null || typeof (o.host ).armed !== "boolean") { return "host (needs armed/armedAt/source)"; } + if (reviveDecisions(o.decisions) === "invalid") return "decisions"; if (o.steeringLog !== undefined && !Array.isArray(o.steeringLog)) return "steeringLog"; return "(unknown)"; } @@ -983,11 +1035,21 @@ function taskDependenciesMet(phase , task ) ); } +export function openDecisionIdsForPhase(plan , wp ) { + // Only a unique, decided target releases a reference. Missing/duplicate ids fail closed. + return [...new Set(wp.awaitsDecision ?? [])].filter((id) => { + const matches = (plan.decisions ?? []).filter((decision) => decision.id === id); + return !(matches.length === 1 && matches[0].status === "decided"); + }); +} + +function workPhaseReadyConditionsMet(plan , wp ) { + return workPhaseDependenciesMet(plan, wp) && openDecisionIdsForPhase(plan, wp).length === 0; +} + function isRunnablePhase(plan , wp ) { - return ( - (wp.status === "pending" || wp.status === "in_progress") - && workPhaseDependenciesMet(plan, wp) - ); + return (wp.status === "pending" || wp.status === "in_progress") + && workPhaseReadyConditionsMet(plan, wp); } export function readyWorkPhases(plan ) { @@ -1060,6 +1122,8 @@ export function dependencyWaitReasons(plan ) { unmetPhaseDependencies.map((id) => describePhaseDependency(plan, id)), )); } + const decisionIds = openDecisionIdsForPhase(plan, wp); + if (decisionIds.length > 0) reasons.push(`work-phase ${wp.id} awaits decision ${decisionIds.join(", ")}`); if (wp.status !== "pending" && wp.status !== "in_progress") continue; for (const task of wp.tasks.filter((candidate) => candidate.status === "pending")) { const unmetTaskDependencies = unmetTaskDependencyIds(wp, task); @@ -1097,6 +1161,8 @@ export function dependencyDeadlock(plan ) { reasons.push( `work-phase ${wp.id} is blocked${wp.blockedReason ? ` (${wp.blockedReason})` : ""}`, ); + const decisionIds = openDecisionIdsForPhase(plan, wp); + if (decisionIds.length > 0) reasons.push(`work-phase ${wp.id} awaits decision ${decisionIds.join(", ")}`); continue; } const unmetPhaseDependencies = unmetPhaseDependencyIds(plan, wp); @@ -1105,8 +1171,10 @@ export function dependencyDeadlock(plan ) { `work-phase ${wp.id}`, unmetPhaseDependencies.map((id) => describePhaseDependency(plan, id)), )); - continue; } + const decisionIds = openDecisionIdsForPhase(plan, wp); + if (decisionIds.length > 0) reasons.push(`work-phase ${wp.id} awaits decision ${decisionIds.join(", ")}`); + if (unmetPhaseDependencies.length > 0) continue; for (const task of wp.tasks.filter((candidate) => candidate.status === "pending")) { const unmetTaskDependencies = unmetTaskDependencyIds(wp, task); if (unmetTaskDependencies.length > 0) { @@ -1120,6 +1188,31 @@ export function dependencyDeadlock(plan ) { return reasons.length > 0 ? { reasons } : null; } +/** IDLE can yield only when actual open user decisions account for all remaining work. */ +export function remainingWorkAwaitsDecisions(plan ) { + const remaining = remainingWorkPhases(plan); + if (remaining.length === 0) return false; + const byId = new Map(plan.workPhases.map((phase) => [phase.id, phase])); + const waiting = (phase , visiting ) => { + if (phase.status !== "pending" && phase.status !== "in_progress") return false; + if ((phase.awaitsDecision ?? []).some((id) => { + const matches = (plan.decisions ?? []).filter((decision) => decision.id === id); + return matches.length === 1 && matches[0].status === "open"; + })) return true; + if (visiting.has(phase.id)) return false; + visiting.add(phase.id); + const result = (phase.dependsOn ?? []).some((id) => { + const dependency = byId.get(id); + return dependency !== undefined && dependency.status !== "done" && waiting(dependency, visiting); + }); + visiting.delete(phase.id); + return result; + }; + const waitingPhases = remaining.filter((phase) => waiting(phase, new Set())); + if (waitingPhases.length !== remaining.length) return false; + return unmetCriteria(plan).every((criterion) => waitingPhases.some((phase) => phase.criteriaIds.includes(criterion.id))); +} + const LIFECYCLE_ID_RE = /^[a-z0-9][a-z0-9-]{0,39}$/; @@ -1127,6 +1220,55 @@ const LIFECYCLE_ID_RE = /^[a-z0-9][a-z0-9-]{0,39}$/; +export function askGoalplanDecision( + plan , + input , +) { + const id = input.id.trim(); + const question = input.question.trim(); + const recommendation = input.recommendation?.trim(); + const workPhaseIds = input.workPhaseIds.map((phaseId) => phaseId.trim()); + if (!LIFECYCLE_ID_RE.test(id)) return { kind: "rejected", reason: "decision id must be a short lowercase id, e.g. dec-1" }; + if (!question) return { kind: "rejected", reason: "decision question must not be empty" }; + if (input.recommendation !== undefined && !recommendation) return { kind: "rejected", reason: "decision recommendation must not be empty" }; + if (!validIsoTime(input.askedAt)) return { kind: "rejected", reason: "decision askedAt must be an ISO timestamp" }; + if (plan.decisions?.some((decision) => decision.id === id)) return { kind: "rejected", reason: `decision '${id}' is already in this plan` }; + const duplicate = plan.decisions?.find((decision) => decision.status === "open" && decision.question.trim() === question); + if (duplicate) return { kind: "rejected", reason: `question is already open as decision '${duplicate.id}'` }; + if (workPhaseIds.some((phaseId) => !phaseId) || new Set(workPhaseIds).size !== workPhaseIds.length) { + return { kind: "rejected", reason: "--work-phase requires distinct non-empty ids" }; + } + for (const phaseId of workPhaseIds) { + const phase = plan.workPhases.find((wp) => wp.id === phaseId); + if (!phase) return { kind: "rejected", reason: `work phase '${phaseId}' is not in this plan` }; + if (phase.status === "done" || phase.status === "superseded") { + return { kind: "rejected", reason: `work phase '${phaseId}' is ${phase.status} and cannot await a decision` }; + } + } + const decision = { id, question, status: "open", askedAt: input.askedAt, + ...(recommendation === undefined ? {} : { recommendation }) }; + const next = { ...plan, decisions: [...(plan.decisions ?? []), decision], + workPhases: plan.workPhases.map((wp) => workPhaseIds.includes(wp.id) + ? { ...wp, awaitsDecision: [...(wp.awaitsDecision ?? []), id] } : wp) }; + const reasons = goalplanDefinitionIntegrityReasons(next); + return reasons.length ? { kind: "rejected", reason: reasons.join("; ") } : { kind: "changed", plan: next }; +} + +export function decideGoalplanDecision( + plan , id , answer , decidedAt , +) { + const decision = plan.decisions?.find((candidate) => candidate.id === id.trim()); + if (!decision) return { kind: "rejected", reason: `decision '${id.trim()}' is not in this plan` }; + if (!answer.trim()) return { kind: "rejected", reason: "decision answer must not be empty" }; + if (!validIsoTime(decidedAt)) return { kind: "rejected", reason: "decision decidedAt must be an ISO timestamp" }; + if (decision.status === "decided") return decision.answer === answer.trim() + ? { kind: "unchanged", plan, reason: `decision '${id.trim()}' is already decided` } + : { kind: "rejected", reason: `decision '${id.trim()}' already has a different answer` }; + const next = { ...plan, decisions: plan.decisions .map((candidate) => candidate.id === decision.id + ? { ...candidate, status: "decided" , answer: answer.trim(), decidedAt } : candidate) }; + return { kind: "changed", plan: next }; +} + export function addGoalplanTask( plan , workPhaseId , @@ -1317,6 +1459,22 @@ export function goalplanDefinitionIntegrityReasons(plan ) { for (const id of duplicateIds(plan.workPhases.map((phase) => phase.id))) { reasons.push(`duplicate work phase id '${id}' makes dependency references ambiguous`); } + const decisionsById = new Map((plan.decisions ?? []).map((decision) => [decision.id, decision])); + for (const id of duplicateIds((plan.decisions ?? []).map((decision) => decision.id))) { + reasons.push(`duplicate decision id '${id}' makes awaitsDecision references ambiguous`); + } + for (const phase of plan.workPhases) { + for (const id of duplicateIds(phase.awaitsDecision ?? [])) { + reasons.push(`work phase ${phase.id} awaits decision '${id}' more than once`); + } + for (const id of new Set(phase.awaitsDecision ?? [])) { + const decision = decisionsById.get(id); + if (!decision) reasons.push(`work phase ${phase.id} awaits unknown decision '${id}'`); + else if (phase.status === "done" && decision.status === "open") { + reasons.push(`work phase ${phase.id} is done while decision ${id} is open`); + } + } + } for (const phase of plan.workPhases) { // 감사 라운드 1 BLOCKER 1: 같은 참조를 여러 번 쓴 dependsOn이 같은 사유를 반복하면 // goal-gate의 slice(0, 4)가 한 문장으로 네 칸을 채워 다른 진단을 가린다. wp2 reviver는 @@ -1795,8 +1953,11 @@ export function closeFixedWorkPhase( // status alone let a target through whose dependency turned blocked after the // marker was written — advanceWorkPhase() answers no_active there, and recovery // must not answer ok. - if (!workPhaseDependenciesMet(plan, current)) { - return { kind: "dependencies_unmet", unmet: unmetPhaseDependencyIds(plan, current) }; + if (!workPhaseReadyConditionsMet(plan, current)) { + return { kind: "dependencies_unmet", unmet: [ + ...unmetPhaseDependencyIds(plan, current), + ...openDecisionIdsForPhase(plan, current).map((id) => `decision:${id}`), + ] }; } // CYCLE-COMPLETION-01, unchanged wording and unchanged variant: an open task keeps @@ -1823,10 +1984,10 @@ export function closeFixedWorkPhase( let next ; if (recordedNext === undefined) { const after = closedWorkPhases.slice(currentIdx + 1).find( - (wp) => wp.status === "pending" && workPhaseDependenciesMet(closedPlan, wp), + (wp) => wp.status === "pending" && isRunnablePhase(closedPlan, wp), ); next = after ?? closedWorkPhases.slice(0, currentIdx).find( - (wp) => wp.status === "pending" && workPhaseDependenciesMet(closedPlan, wp), + (wp) => wp.status === "pending" && isRunnablePhase(closedPlan, wp), ); } else if (recordedNext === null) { next = undefined; @@ -1871,11 +2032,11 @@ export function closeFixedWorkPhase( // and it can only do that against a normalized cursor. next = closedWorkPhases.find( (wp) => wp.id === plan.activeWorkPhaseId && wp.id !== workPhaseId - && wp.status === "in_progress" && workPhaseDependenciesMet(closedPlan, wp), + && wp.status === "in_progress" && workPhaseReadyConditionsMet(closedPlan, wp), ); } else if (named.status !== "pending" && named.status !== "in_progress") { return { kind: "successor_lost", successorId: recordedNext, reason: "not_runnable" }; - } else if (!workPhaseDependenciesMet(closedPlan, named)) { + } else if (!workPhaseReadyConditionsMet(closedPlan, named)) { return { kind: "successor_lost", successorId: recordedNext, reason: "dependencies_unmet" }; } else { next = named; @@ -1925,7 +2086,7 @@ export function absentSuccessorDetail( ? "is gone too" : reason === "not_runnable" ? "can no longer be started" - : "now waits for another work-phase"; + : "now waits for a prerequisite or decision"; } @@ -1954,7 +2115,7 @@ export function resumeAbsentTarget( // successor waiting on the same unmet dependency was refused with the target present and // activated with it gone. A dangling dependsOn reads as not-done here by design, and the // pending branch already refused that plan. - if (!workPhaseDependenciesMet(plan, named)) { + if (!workPhaseReadyConditionsMet(plan, named)) { return { kind: "successor_lost", successorId: recordedNext, reason: "dependencies_unmet" }; } // Running: the activation happened too, but only if the cursor agrees. §45 established @@ -2044,11 +2205,11 @@ export function effectiveActiveWorkPhaseId(plan ) { if (cur && isRunnablePhase(plan, cur)) return cur.id; } const inProgress = plan.workPhases.find( - (wp) => wp.status === "in_progress" && workPhaseDependenciesMet(plan, wp), + (wp) => wp.status === "in_progress" && isRunnablePhase(plan, wp), ); if (inProgress) return inProgress.id; const pending = plan.workPhases.find( - (wp) => wp.status === "pending" && workPhaseDependenciesMet(plan, wp), + (wp) => wp.status === "pending" && isRunnablePhase(plan, wp), ); return pending?.id ?? null; } diff --git a/plugins/codexclaw/components/pabcd-state/dist/hook.js b/plugins/codexclaw/components/pabcd-state/dist/hook.js index d47d7258..8ce42ed1 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/hook.js +++ b/plugins/codexclaw/components/pabcd-state/dist/hook.js @@ -74,6 +74,7 @@ import { readGoalplan, readyTasks, readyWorkPhases, + remainingWorkAwaitsDecisions, unmetCriteria, withGoalplanWriteLock, writeGoalplan, @@ -1822,7 +1823,9 @@ export function handleStop( // plan gets a bounded arming block — "IDLE is not the end while work remains". if (!inFlight) { if (!goalActive) return ""; - if (!state.slug || !safeReadBoundGoalplan(payload.cwd, state.slug)) return ""; + const plan = state.slug ? safeReadBoundGoalplan(payload.cwd, state.slug) : null; + if (!plan) return ""; + if (remainingWorkAwaitsDecisions(plan)) return ""; // bail: don't pile on during context-pressure/compaction recovery. if (isContextPressureTail(readTranscriptTail(payload.transcript_path))) return ""; const count = bumpStopCounter(payload.cwd, state); From c715dbd92145c5ddef7b5f68a80c64a66df65220 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 03:15:44 +0900 Subject: [PATCH 60/90] docs: publish measured test count (3704) --- README.ko.md | 2 +- README.md | 2 +- README.zh.md | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/README.ko.md b/README.ko.md index 82a189fe..1d09231d 100644 --- a/README.ko.md +++ b/README.ko.md @@ -13,7 +13,7 @@

CI - 3,680 tests + 3,704 tests 29 skills 31 hooks Documentation diff --git a/README.md b/README.md index 5560778a..1857db65 100644 --- a/README.md +++ b/README.md @@ -13,7 +13,7 @@

CI - 3,680 tests + 3,704 tests 29 skills 31 hooks Documentation diff --git a/README.zh.md b/README.zh.md index cd7ae04d..0bbc6829 100644 --- a/README.zh.md +++ b/README.zh.md @@ -13,7 +13,7 @@

CI - 3,680 tests + 3,704 tests 29 skills 31 hooks Documentation From 4d6dfb5cbba6f9c17bf092806186244e341a946c Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 03:16:01 +0900 Subject: [PATCH 61/90] docs(changelog): goalplan pending decisions (#262) --- CHANGELOG.md | 1 + 1 file changed, 1 insertion(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 988480d9..0c6d0248 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -15,6 +15,7 @@ All notable changes to codexclaw are documented here. The format follows ### Changed +- Goalplans can record pending user decisions: `cxc loop ask --session --id --question [--recommendation ] [--work-phase ]...` links a question the agent already asked to the phases that wait on it, and `cxc loop decide --session --id --answer ` records the answer. Linked phases are not runnable while the decision is open; unrelated phases stay ready. When every remaining phase and unmet criterion waits on an open decision, the Stop hook lets an IDLE turn end instead of asking to start another phase; the goal stays active and cannot be completed early. Old plans load unchanged (#262). - The absolute Stop continuation cap (24) now counts per genuine user turn instead of per session, and the release prints one notice per turn (#254). ### Fixed From f28994d1351ccde5bc507f14724d170f49aa1aad Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 03:22:08 +0900 Subject: [PATCH 62/90] fix(pabcd-state): decide refuses ambiguous ids; IDLE release requires an intact plan (#262) --- .../components/pabcd-state/dist/goalplan.js | 8 ++++++-- .../components/pabcd-state/src/goalplan.ts | 8 ++++++-- .../test/work-phase-states.test.ts | 20 +++++++++++++++++++ 3 files changed, 32 insertions(+), 4 deletions(-) diff --git a/plugins/codexclaw/components/pabcd-state/dist/goalplan.js b/plugins/codexclaw/components/pabcd-state/dist/goalplan.js index a20ac94b..1f6d4702 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/goalplan.js +++ b/plugins/codexclaw/components/pabcd-state/dist/goalplan.js @@ -1190,6 +1190,8 @@ export function dependencyDeadlock(plan ) { /** IDLE can yield only when actual open user decisions account for all remaining work. */ export function remainingWorkAwaitsDecisions(plan ) { + // A plan with broken references must keep prompting the agent to repair it. + if (goalplanDefinitionIntegrityReasons(plan).length > 0) return false; const remaining = remainingWorkPhases(plan); if (remaining.length === 0) return false; const byId = new Map(plan.workPhases.map((phase) => [phase.id, phase])); @@ -1257,8 +1259,10 @@ export function askGoalplanDecision( export function decideGoalplanDecision( plan , id , answer , decidedAt , ) { - const decision = plan.decisions?.find((candidate) => candidate.id === id.trim()); - if (!decision) return { kind: "rejected", reason: `decision '${id.trim()}' is not in this plan` }; + const matches = (plan.decisions ?? []).filter((candidate) => candidate.id === id.trim()); + if (matches.length === 0) return { kind: "rejected", reason: `decision '${id.trim()}' is not in this plan` }; + if (matches.length > 1) return { kind: "rejected", reason: `decision id '${id.trim()}' is ambiguous (${matches.length} entries); repair the plan first` }; + const decision = matches[0]; if (!answer.trim()) return { kind: "rejected", reason: "decision answer must not be empty" }; if (!validIsoTime(decidedAt)) return { kind: "rejected", reason: "decision decidedAt must be an ISO timestamp" }; if (decision.status === "decided") return decision.answer === answer.trim() diff --git a/plugins/codexclaw/components/pabcd-state/src/goalplan.ts b/plugins/codexclaw/components/pabcd-state/src/goalplan.ts index 4bc77b96..cc67699f 100644 --- a/plugins/codexclaw/components/pabcd-state/src/goalplan.ts +++ b/plugins/codexclaw/components/pabcd-state/src/goalplan.ts @@ -1190,6 +1190,8 @@ export function dependencyDeadlock(plan: Goalplan): DependencyDeadlock | null { /** IDLE can yield only when actual open user decisions account for all remaining work. */ export function remainingWorkAwaitsDecisions(plan: Goalplan): boolean { + // A plan with broken references must keep prompting the agent to repair it. + if (goalplanDefinitionIntegrityReasons(plan).length > 0) return false; const remaining = remainingWorkPhases(plan); if (remaining.length === 0) return false; const byId = new Map(plan.workPhases.map((phase) => [phase.id, phase])); @@ -1257,8 +1259,10 @@ export function askGoalplanDecision( export function decideGoalplanDecision( plan: Goalplan, id: string, answer: string, decidedAt: string, ): GoalplanLifecycleResult { - const decision = plan.decisions?.find((candidate) => candidate.id === id.trim()); - if (!decision) return { kind: "rejected", reason: `decision '${id.trim()}' is not in this plan` }; + const matches = (plan.decisions ?? []).filter((candidate) => candidate.id === id.trim()); + if (matches.length === 0) return { kind: "rejected", reason: `decision '${id.trim()}' is not in this plan` }; + if (matches.length > 1) return { kind: "rejected", reason: `decision id '${id.trim()}' is ambiguous (${matches.length} entries); repair the plan first` }; + const decision = matches[0]; if (!answer.trim()) return { kind: "rejected", reason: "decision answer must not be empty" }; if (!validIsoTime(decidedAt)) return { kind: "rejected", reason: "decision decidedAt must be an ISO timestamp" }; if (decision.status === "decided") return decision.answer === answer.trim() diff --git a/plugins/codexclaw/components/pabcd-state/test/work-phase-states.test.ts b/plugins/codexclaw/components/pabcd-state/test/work-phase-states.test.ts index babbe36a..f4171b36 100644 --- a/plugins/codexclaw/components/pabcd-state/test/work-phase-states.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/work-phase-states.test.ts @@ -15,6 +15,7 @@ import { advanceWorkPhase, buildGoalplan, closeFixedWorkPhase, + decideGoalplanDecision, resumeAbsentTarget, goalplanDefinitionIntegrityReasons, readGoalplanDetailed, @@ -445,3 +446,22 @@ test("remaining work awaits decisions only for actual open answers and covered c assert.equal(remainingWorkAwaitsDecisions({ ...p, criteria: [...p.criteria, { ...criterion, id: "c-2" }] }), false); assert.equal(remainingWorkAwaitsDecisions({ ...p, workPhases: [...p.workPhases, phase("free", "pending")] }), false); }); + + +test("IDLE release refuses a plan with broken references", () => { + const d = { id: "dec-1", question: "Choose", status: "open" as const, askedAt: "2026-09-28T00:00:00.000Z" }; + const waiting = phase("root", "in_progress", { awaitsDecision: ["dec-1"] }); + const base = plan([waiting], { decisions: [d] }); + assert.equal(remainingWorkAwaitsDecisions(base), true); + assert.equal(remainingWorkAwaitsDecisions(plan([phase("root", "in_progress", { awaitsDecision: ["dec-1", "ghost"] })], { decisions: [d] })), false); + assert.equal(remainingWorkAwaitsDecisions(plan([waiting, phase("child", "pending", { dependsOn: ["missing"] })], { decisions: [d] })), false); +}); + +test("decide refuses an ambiguous decision id and leaves the plan unchanged", () => { + const a = { id: "dec-1", question: "First", status: "open" as const, askedAt: "2026-09-28T00:00:00.000Z" }; + const b = { id: "dec-1", question: "Second", status: "open" as const, askedAt: "2026-09-28T00:00:00.000Z" }; + const p = plan([phase("root", "pending", { awaitsDecision: ["dec-1"] })], { decisions: [a, b] }); + const result = decideGoalplanDecision(p, "dec-1", "yes", "2026-09-28T01:00:00.000Z"); + assert.equal(result.kind, "rejected"); + assert.match((result as { reason: string }).reason, /ambiguous/); +}); From cc5ce06910bd583dec47b95b51c6c8af7fb207e9 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 03:22:08 +0900 Subject: [PATCH 63/90] docs: publish measured test count (3706) --- README.ko.md | 2 +- README.md | 2 +- README.zh.md | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/README.ko.md b/README.ko.md index 1d09231d..49610d5b 100644 --- a/README.ko.md +++ b/README.ko.md @@ -13,7 +13,7 @@

CI - 3,704 tests + 3,706 tests 29 skills 31 hooks Documentation diff --git a/README.md b/README.md index 1857db65..850f8494 100644 --- a/README.md +++ b/README.md @@ -13,7 +13,7 @@

CI - 3,704 tests + 3,706 tests 29 skills 31 hooks Documentation diff --git a/README.zh.md b/README.zh.md index 0bbc6829..8244581b 100644 --- a/README.zh.md +++ b/README.zh.md @@ -13,7 +13,7 @@

CI - 3,704 tests + 3,706 tests 29 skills 31 hooks Documentation From 5cb377deb40df9e6e3f423c783e29e8f60a775ab Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 03:23:55 +0900 Subject: [PATCH 64/90] fix(pabcd-state): IDLE release also requires dependency-completion integrity (#262) --- plugins/codexclaw/components/pabcd-state/dist/goalplan.js | 3 ++- plugins/codexclaw/components/pabcd-state/src/goalplan.ts | 3 ++- .../components/pabcd-state/test/work-phase-states.test.ts | 7 +++++++ 3 files changed, 11 insertions(+), 2 deletions(-) diff --git a/plugins/codexclaw/components/pabcd-state/dist/goalplan.js b/plugins/codexclaw/components/pabcd-state/dist/goalplan.js index 1f6d4702..6220dfb4 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/goalplan.js +++ b/plugins/codexclaw/components/pabcd-state/dist/goalplan.js @@ -1191,7 +1191,8 @@ export function dependencyDeadlock(plan ) { /** IDLE can yield only when actual open user decisions account for all remaining work. */ export function remainingWorkAwaitsDecisions(plan ) { // A plan with broken references must keep prompting the agent to repair it. - if (goalplanDefinitionIntegrityReasons(plan).length > 0) return false; + if (goalplanDefinitionIntegrityReasons(plan).length > 0 || + goalplanDependencyCompletionReasons(plan).length > 0) return false; const remaining = remainingWorkPhases(plan); if (remaining.length === 0) return false; const byId = new Map(plan.workPhases.map((phase) => [phase.id, phase])); diff --git a/plugins/codexclaw/components/pabcd-state/src/goalplan.ts b/plugins/codexclaw/components/pabcd-state/src/goalplan.ts index cc67699f..9d7eb33d 100644 --- a/plugins/codexclaw/components/pabcd-state/src/goalplan.ts +++ b/plugins/codexclaw/components/pabcd-state/src/goalplan.ts @@ -1191,7 +1191,8 @@ export function dependencyDeadlock(plan: Goalplan): DependencyDeadlock | null { /** IDLE can yield only when actual open user decisions account for all remaining work. */ export function remainingWorkAwaitsDecisions(plan: Goalplan): boolean { // A plan with broken references must keep prompting the agent to repair it. - if (goalplanDefinitionIntegrityReasons(plan).length > 0) return false; + if (goalplanDefinitionIntegrityReasons(plan).length > 0 || + goalplanDependencyCompletionReasons(plan).length > 0) return false; const remaining = remainingWorkPhases(plan); if (remaining.length === 0) return false; const byId = new Map(plan.workPhases.map((phase) => [phase.id, phase])); diff --git a/plugins/codexclaw/components/pabcd-state/test/work-phase-states.test.ts b/plugins/codexclaw/components/pabcd-state/test/work-phase-states.test.ts index f4171b36..ff314171 100644 --- a/plugins/codexclaw/components/pabcd-state/test/work-phase-states.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/work-phase-states.test.ts @@ -465,3 +465,10 @@ test("decide refuses an ambiguous decision id and leaves the plan unchanged", () assert.equal(result.kind, "rejected"); assert.match((result as { reason: string }).reason, /ambiguous/); }); + + +test("IDLE release refuses a plan whose done phase depends on unfinished work", () => { + const d = { id: "dec-1", question: "Choose", status: "open" as const, askedAt: "2026-09-28T00:00:00.000Z" }; + const p = plan([phase("root", "pending", { awaitsDecision: ["dec-1"] }), phase("child", "done", { dependsOn: ["root"] })], { decisions: [d] }); + assert.equal(remainingWorkAwaitsDecisions(p), false); +}); From 40a490dc15f014b0c5b2231fc1637c3af7f3d688 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 03:23:55 +0900 Subject: [PATCH 65/90] docs: publish measured test count (3707) --- README.ko.md | 2 +- README.md | 2 +- README.zh.md | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/README.ko.md b/README.ko.md index 49610d5b..ed7d27b8 100644 --- a/README.ko.md +++ b/README.ko.md @@ -13,7 +13,7 @@

CI - 3,706 tests + 3,707 tests 29 skills 31 hooks Documentation diff --git a/README.md b/README.md index 850f8494..600030c6 100644 --- a/README.md +++ b/README.md @@ -13,7 +13,7 @@

CI - 3,706 tests + 3,707 tests 29 skills 31 hooks Documentation diff --git a/README.zh.md b/README.zh.md index 8244581b..d2197d63 100644 --- a/README.zh.md +++ b/README.zh.md @@ -13,7 +13,7 @@

CI - 3,706 tests + 3,707 tests 29 skills 31 hooks Documentation From 04916fd5764b3a0e317fb8810ef1fc81bf6eb88b Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 03:26:02 +0900 Subject: [PATCH 66/90] fix(pabcd-state): IDLE decision release refuses every structural E8 reason (#262) --- .../components/pabcd-state/dist/goalplan.js | 22 +++++++++++++++++-- .../components/pabcd-state/src/goalplan.ts | 22 +++++++++++++++++-- .../test/work-phase-states.test.ts | 9 ++++++++ 3 files changed, 49 insertions(+), 4 deletions(-) diff --git a/plugins/codexclaw/components/pabcd-state/dist/goalplan.js b/plugins/codexclaw/components/pabcd-state/dist/goalplan.js index 6220dfb4..c03c80a4 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/goalplan.js +++ b/plugins/codexclaw/components/pabcd-state/dist/goalplan.js @@ -1191,8 +1191,7 @@ export function dependencyDeadlock(plan ) { /** IDLE can yield only when actual open user decisions account for all remaining work. */ export function remainingWorkAwaitsDecisions(plan ) { // A plan with broken references must keep prompting the agent to repair it. - if (goalplanDefinitionIntegrityReasons(plan).length > 0 || - goalplanDependencyCompletionReasons(plan).length > 0) return false; + if (goalplanStructuralReasons(plan).length > 0) return false; const remaining = remainingWorkPhases(plan); if (remaining.length === 0) return false; const byId = new Map(plan.workPhases.map((phase) => [phase.id, phase])); @@ -1708,6 +1707,25 @@ function supersededIntegrityReasons(plan ) { return out; } +/** + * Every E8 reason that means the plan itself is broken, as opposed to work that is + * simply not finished yet. The IDLE decision release refuses any of these so the + * agent keeps being prompted to repair the plan. + */ +export function goalplanStructuralReasons(plan ) { + const reasons = []; + if (typeof plan.schemaVersion === "number" && plan.schemaVersion > SUPPORTED_MAX_SCHEMA_VERSION) { + reasons.push(`schemaVersion ${plan.schemaVersion} is newer than this build supports`); + } + reasons.push(...goalplanDefinitionIntegrityReasons(plan), ...goalplanDependencyCompletionReasons(plan)); + for (const c of plan.criteria) { + if (c.status === "met" && (c.capturedEvidence ?? "").trim().length === 0) reasons.push(`criterion ${c.id} marked met but has no captured evidence`); + } + for (const wp of doneWorkPhasesWithPendingTasks(plan)) reasons.push(`work phase ${wp.id} is marked done but still has open task(s)`); + reasons.push(...supersededIntegrityReasons(plan)); + return reasons; +} + /** Marker path: promotion to v2 is recorded outside the plan file as well. */ export function schemaMarkerPath(cwd , slug ) { return join(goalplanDir(cwd, slug), "schema-v2.marker"); diff --git a/plugins/codexclaw/components/pabcd-state/src/goalplan.ts b/plugins/codexclaw/components/pabcd-state/src/goalplan.ts index 9d7eb33d..92bc5e20 100644 --- a/plugins/codexclaw/components/pabcd-state/src/goalplan.ts +++ b/plugins/codexclaw/components/pabcd-state/src/goalplan.ts @@ -1191,8 +1191,7 @@ export function dependencyDeadlock(plan: Goalplan): DependencyDeadlock | null { /** IDLE can yield only when actual open user decisions account for all remaining work. */ export function remainingWorkAwaitsDecisions(plan: Goalplan): boolean { // A plan with broken references must keep prompting the agent to repair it. - if (goalplanDefinitionIntegrityReasons(plan).length > 0 || - goalplanDependencyCompletionReasons(plan).length > 0) return false; + if (goalplanStructuralReasons(plan).length > 0) return false; const remaining = remainingWorkPhases(plan); if (remaining.length === 0) return false; const byId = new Map(plan.workPhases.map((phase) => [phase.id, phase])); @@ -1708,6 +1707,25 @@ function supersededIntegrityReasons(plan: Goalplan): string[] { return out; } +/** + * Every E8 reason that means the plan itself is broken, as opposed to work that is + * simply not finished yet. The IDLE decision release refuses any of these so the + * agent keeps being prompted to repair the plan. + */ +export function goalplanStructuralReasons(plan: Goalplan): string[] { + const reasons: string[] = []; + if (typeof plan.schemaVersion === "number" && plan.schemaVersion > SUPPORTED_MAX_SCHEMA_VERSION) { + reasons.push(`schemaVersion ${plan.schemaVersion} is newer than this build supports`); + } + reasons.push(...goalplanDefinitionIntegrityReasons(plan), ...goalplanDependencyCompletionReasons(plan)); + for (const c of plan.criteria) { + if (c.status === "met" && (c.capturedEvidence ?? "").trim().length === 0) reasons.push(`criterion ${c.id} marked met but has no captured evidence`); + } + for (const wp of doneWorkPhasesWithPendingTasks(plan)) reasons.push(`work phase ${wp.id} is marked done but still has open task(s)`); + reasons.push(...supersededIntegrityReasons(plan)); + return reasons; +} + /** Marker path: promotion to v2 is recorded outside the plan file as well. */ export function schemaMarkerPath(cwd: string, slug: string): string { return join(goalplanDir(cwd, slug), "schema-v2.marker"); diff --git a/plugins/codexclaw/components/pabcd-state/test/work-phase-states.test.ts b/plugins/codexclaw/components/pabcd-state/test/work-phase-states.test.ts index ff314171..97348cfa 100644 --- a/plugins/codexclaw/components/pabcd-state/test/work-phase-states.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/work-phase-states.test.ts @@ -472,3 +472,12 @@ test("IDLE release refuses a plan whose done phase depends on unfinished work", const p = plan([phase("root", "pending", { awaitsDecision: ["dec-1"] }), phase("child", "done", { dependsOn: ["root"] })], { decisions: [d] }); assert.equal(remainingWorkAwaitsDecisions(p), false); }); + + +test("IDLE release refuses invalid superseded phases and other structural breaks", () => { + const d = { id: "dec-1", question: "Choose", status: "open" as const, askedAt: "2026-09-28T00:00:00.000Z" }; + const waiting = phase("root", "pending", { awaitsDecision: ["dec-1"] }); + assert.equal(remainingWorkAwaitsDecisions(plan([waiting, phase("old", "superseded", { supersededBy: "ghost" })], { decisions: [d] })), false); + assert.equal(remainingWorkAwaitsDecisions(plan([waiting, phase("gone", "superseded")], { decisions: [d] })), false); + assert.equal(remainingWorkAwaitsDecisions(plan([waiting, phase("old", "superseded", { supersededBy: "root" })], { decisions: [d] })), true); +}); From e59c9d8ee825e60eccdfdd0e71e75be7e18b2d11 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 03:26:02 +0900 Subject: [PATCH 67/90] docs: publish measured test count (3708) --- README.ko.md | 2 +- README.md | 2 +- README.zh.md | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/README.ko.md b/README.ko.md index ed7d27b8..d2965583 100644 --- a/README.ko.md +++ b/README.ko.md @@ -13,7 +13,7 @@

CI - 3,707 tests + 3,708 tests 29 skills 31 hooks Documentation diff --git a/README.md b/README.md index 600030c6..275275d8 100644 --- a/README.md +++ b/README.md @@ -13,7 +13,7 @@

CI - 3,707 tests + 3,708 tests 29 skills 31 hooks Documentation diff --git a/README.zh.md b/README.zh.md index d2197d63..0d721a8f 100644 --- a/README.zh.md +++ b/README.zh.md @@ -13,7 +13,7 @@

CI - 3,707 tests + 3,708 tests 29 skills 31 hooks Documentation From 959cb4d7bce49bc779b425e014cef8363a99d2ec Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 03:30:26 +0900 Subject: [PATCH 68/90] docs(plan): wp5 executable delivery plan --- devlog/_plan/260927_issue_train/040_wp5_delivery.md | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/devlog/_plan/260927_issue_train/040_wp5_delivery.md b/devlog/_plan/260927_issue_train/040_wp5_delivery.md index e3766299..6795c440 100644 --- a/devlog/_plan/260927_issue_train/040_wp5_delivery.md +++ b/devlog/_plan/260927_issue_train/040_wp5_delivery.md @@ -26,3 +26,16 @@ This phase lands nothing new; it proves the merged state and records the issue d ## Acceptance All goalplan criteria met with captured evidence; `cxc loop validate` passes; `origin/dev` contains the three merge commits; no PR into `main` was opened by this unit. + + +## wp5 executable plan (2026-09-28) + +State at entry: PR #269 (wp2) merged as b558d84e and PR #270 (wp3) merged as c550d78e, both after all 13 checks passed on their heads; PR #271 (wp4) open at the codex/issue-train-wp4 head. + +1. Wait for PR #271's hosted CI on its head SHA; merge with a merge commit (same method as #269 and #270, so later branches stay clean) using `--match-head-commit`. +2. Close the fixed issues with a comment that names the merged PR: #250, #251, #253 and #252, #254 (implemented) through #269; #262 through #271. #255 gets a comment that the `.gitignore` half shipped in #269 and lazy session-state creation is deferred (reason in 001), and it stays open. +3. Comment on each deferred issue (#209, #213, #247, #256, #257, #258, #259, #260, #263, #264, #266, #267, #268) with the one-line reason from 001 and a link to `devlog/_plan/260927_issue_train/001_research.md` on dev; they stay open. #261 gets the decline reason and is closed as not planned. #265 gets a comment linking the guidance in #270 and stays open for the deferred CLI. +4. Write `041_delivery.md` (PRs, merge commits, CI run evidence, issue dispositions, residual risks, the plugin-cache and log observations) and deliver it through a small docs PR to dev (CI, merge). +5. Verify: every closed issue shows the closing comment; `origin/dev` contains the four merge commits; `cxc loop validate` passes. + +No architect consultation: this phase makes no design decisions (cxc-dev §0.1 fast path for ops and docs); the independent A review covers the disposition list. From 42a18758c24a5bbd11fa139ccedb9e924717a214 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 03:33:36 +0900 Subject: [PATCH 69/90] docs(plan): keep #262 open as partial --- devlog/_plan/260927_issue_train/040_wp5_delivery.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/devlog/_plan/260927_issue_train/040_wp5_delivery.md b/devlog/_plan/260927_issue_train/040_wp5_delivery.md index 6795c440..8af00a07 100644 --- a/devlog/_plan/260927_issue_train/040_wp5_delivery.md +++ b/devlog/_plan/260927_issue_train/040_wp5_delivery.md @@ -33,7 +33,7 @@ All goalplan criteria met with captured evidence; `cxc loop validate` passes; `o State at entry: PR #269 (wp2) merged as b558d84e and PR #270 (wp3) merged as c550d78e, both after all 13 checks passed on their heads; PR #271 (wp4) open at the codex/issue-train-wp4 head. 1. Wait for PR #271's hosted CI on its head SHA; merge with a merge commit (same method as #269 and #270, so later branches stay clean) using `--match-head-commit`. -2. Close the fixed issues with a comment that names the merged PR: #250, #251, #253 and #252, #254 (implemented) through #269; #262 through #271. #255 gets a comment that the `.gitignore` half shipped in #269 and lazy session-state creation is deferred (reason in 001), and it stays open. +2. Close the fixed issues with a comment that names the merged PR: #250, #251, #253 and #252, #254 (implemented) through #269. #262 gets a partial-implementation comment naming what #271 shipped (open/decided decisions, phase links, readiness and Stop behavior) and what it did not (`options[]`, validation that the recommendation is one of them, `withdrawn`); it stays open. #255 gets a comment that the `.gitignore` half shipped in #269 and lazy session-state creation is deferred (reason in 001), and it stays open. 3. Comment on each deferred issue (#209, #213, #247, #256, #257, #258, #259, #260, #263, #264, #266, #267, #268) with the one-line reason from 001 and a link to `devlog/_plan/260927_issue_train/001_research.md` on dev; they stay open. #261 gets the decline reason and is closed as not planned. #265 gets a comment linking the guidance in #270 and stays open for the deferred CLI. 4. Write `041_delivery.md` (PRs, merge commits, CI run evidence, issue dispositions, residual risks, the plugin-cache and log observations) and deliver it through a small docs PR to dev (CI, merge). 5. Verify: every closed issue shows the closing comment; `origin/dev` contains the four merge commits; `cxc loop validate` passes. From 941bf4b0340bab6952a28c6e27fdcbacde50e3b7 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Mon, 28 Sep 2026 03:42:48 +0900 Subject: [PATCH 70/90] docs(plan): issue train delivery record --- .../_plan/260927_issue_train/041_delivery.md | 36 +++++++++++++++++++ 1 file changed, 36 insertions(+) create mode 100644 devlog/_plan/260927_issue_train/041_delivery.md diff --git a/devlog/_plan/260927_issue_train/041_delivery.md b/devlog/_plan/260927_issue_train/041_delivery.md new file mode 100644 index 00000000..5ad56953 --- /dev/null +++ b/devlog/_plan/260927_issue_train/041_delivery.md @@ -0,0 +1,36 @@ +# Delivery record: issue train 2026-09-27 + +Four ordinary PRs landed the train in `dev`: wp2's hook runtime fixes (#269), wp3's agent-created thread permissions and dispatch guidance (#270), wp4's goalplan pending decisions (#271), and this record. Every implementation PR was merged with a merge commit after all hosted checks passed on its exact head, so later branches built on earlier ones stayed clean. Nothing was promoted to `main`, released or tagged. + +## Pull requests + +| PR | Scope | Head at merge | Merge commit | Hosted checks | +|---|---|---|---|---| +| #269 | #250, #251, #252, #253, #254, #255 (partial) | 29e34de8 | b558d84e | 13/13 success (ubuntu, macOS, Windows ×4 shards, packed install, artifacts, labeler, target) | +| #270 | agent-thread permission hook and advisory, dispatch guidance, #265 guidance | 40624cdb | c550d78e | 13/13 success | +| #271 | #262 (partial: open/decided decisions) | e59c9d8e | c4f17670 | 14/14 success | + +Two CI rounds failed on Windows before their fixes: #269's policy-off CLI test built a path from `URL.pathname` (fixed with `fileURLToPath`, 29e34de8), and #270's home-directory test set only `HOME`, which Windows `os.homedir()` ignores (fixed by also setting `USERPROFILE`, 40624cdb). Hosted CI for #269 waited about an hour behind another repository's queued runs on the same account; no other repository's runs were cancelled. + +## Issue dispositions + +- Closed as fixed with a comment naming #269: #250, #251, #252, #253, #254. +- Partly addressed, left open with a comment: #255 (`.gitignore` shipped; lazy state creation deferred), #262 (`options[]`, recommendation validation and `withdrawn` not shipped), #265 (guidance shipped; CLI deferred). +- Deferred with a one-line reason and a link to 001: #209, #213, #247, #256, #257, #258, #259, #260, #263, #264, #266, #267, #268. +- Closed as not planned: #261. + +## Review record + +Each implementation phase had an architect consultation with reflection, a plan audit by the same gpt-6-sol reviewer, parallel gpt-6-sol builders in managed worktrees, and an independent gpt-6-sol implementation review. The implementation reviews found and forced fixes for: negated and indirect refusals still arming the loop (#250), malformed or overflowing TOML and project-controlled config paths authorizing the permission hook, hook-observation writes under a project-contained `CODEX_HOME`, ambiguous decision ids, and the IDLE decision release firing on structurally broken plans. #255's first design (lazy session-state creation) failed three plan audits and was replaced with the `.gitignore` partial fix. + +## Residual risks + +- The #250 detector is advisory and heuristic; unusual refusal phrasing can still produce a hint, never a phase change. +- The permission hook treats explicit top-level `config.toml` keys as evidence of user intent. It does not see runtime overrides or Codex schema-type errors, and live suppression of the Desktop approval modal is source-verified but not yet observed in the app. The two new hooks need trust approval after upgrade. +- `cxc loop ask` cannot prove a question reached the user. + +## Operational notes + +- During this train the installed plugin cache was replaced at 23:45 KST on 2026-09-27 with 0.2.36 from the marketplace source `/Users/jun/Developer/new/700_projects/codexclaw` (checked out at an older commit). The loop used a pinned 0.2.39 CLI extracted from dev for its own FSM commands. The installed plugin was not changed. +- A docs commit on the wp3 branch briefly swept 279 pre-existing untracked files from this checkout into the pushed branch (including a headless Chrome profile used for PDF rendering, which held no cookies or saved logins). The branch was rewritten and force-pushed before merge, so they never reached `dev`; the original commit object may stay reachable on GitHub by SHA for a while. The files remain untracked in the checkout. + From 66669c6ecccfe27f89d2c0f4c0c566b66f211d9f Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Wed, 30 Sep 2026 02:01:36 +0900 Subject: [PATCH 71/90] docs(plan): issue train 0930 roadmap (triage, architect consultation, decade docs) --- devlog/_plan/260930_issue_train/000_plan.md | 72 +++++ .../_plan/260930_issue_train/001_research.md | 32 ++ .../002_architect_consultation.md | 49 +++ .../010_wp2_dispatch_contract.md | 294 ++++++++++++++++++ .../020_wp3_interview_assumptions.md | 128 ++++++++ .../030_wp5_decision_options.md | 161 ++++++++++ .../260930_issue_train/040_wp4_delivery.md | 41 +++ 7 files changed, 777 insertions(+) create mode 100644 devlog/_plan/260930_issue_train/000_plan.md create mode 100644 devlog/_plan/260930_issue_train/001_research.md create mode 100644 devlog/_plan/260930_issue_train/002_architect_consultation.md create mode 100644 devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md create mode 100644 devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md create mode 100644 devlog/_plan/260930_issue_train/030_wp5_decision_options.md create mode 100644 devlog/_plan/260930_issue_train/040_wp4_delivery.md diff --git a/devlog/_plan/260930_issue_train/000_plan.md b/devlog/_plan/260930_issue_train/000_plan.md new file mode 100644 index 00000000..5a37c9cb --- /dev/null +++ b/devlog/_plan/260930_issue_train/000_plan.md @@ -0,0 +1,72 @@ +# Issue train 2026-09-30: dispatch verifier coverage, verifier effects, assumption provenance, decision options, release 0.2.40 + +Codexclaw has one confirmed defect among its 20 open issues: a dispatch receipt can satisfy a packet that requires two verifier commands with one unrelated passing result (#276). Three contained improvements need no maintainer decision and add no hooks: optional verifier write-effect declarations with a pure preflight (#277), provenance for Interview assumptions so inferred ones are not handed to Plan as agreed requirements (#275), and the `options[]` half of goalplan decisions that #271 left out (#262). This unit fixes and ships those, records a decision for every open issue (001), closes the ones that would grow hooks or depend on host signals the plugin cannot see, and releases the result as 0.2.40 from `main`. + +Reader: a maintainer deciding whether to merge these changes and publish 0.2.40; familiarity with the dispatch contract, the Interview skill and the goalplan CLI is assumed. + +## Loop contract + +- Loop archetype: satisfy-spec HOTL, docs-first (LOOP-DOCS-FIRST-01). +- Trigger: the user's request on 2026-09-30 to fix the worthwhile open issues without bloating hooks, run it through cxc-loop, merge to `dev` and `main`, release, close issues at the agent's judgment, and verify each phase against the PABCD initiative rules with parallel inherited-model subagents. +- Goal: `#276`, `#277`, `#275` fixed and closed; `#262` options shipped; every open issue dispositioned; v0.2.40 published from `main` with verified assets. +- Non-goals: new hooks or hook injections; Codex core or Desktop changes; other repositories; #273's CLI verb; #274's schema (maintainer placement decision); #262 `withdrawn`; the installed plugin cache and SSH hosts. +- Verifier: per phase the focused tests in each decade doc, then `npm run build`, focused tests through `cxc receipt test`, `npm test` (TAP total), `inventory.mjs --check --tests `, `gate.mjs`, `platform-smoke.mjs`; hosted CI on each PR head; for the release, `check-versions.mjs 0.2.40`, main exact-SHA CI, the `release.yml` dry run and publish, `shasum -a 256 -c SHA256SUMS` and payload comparison with `git archive`. Skill prose is read by no test; its review is human (PLAN-VERIFIER-REAL-01). +- Stop condition: all goalplan criteria met with fresh evidence, or a real blocker after root-cause work. +- Memory artifact: this unit, the goalplan `.codexclaw/goalplans/codexclaw-issue-train-2026-09-30-repo-lidge-jun/`, and `.codexclaw/evidence/01a0ee0f-ea9e-7273-91fe-0c4188c2cae6/`. +- Expected terminal outcomes: DONE (merged, released, dispositioned); BLOCKED (CI infrastructure, branch protection or credentials outside scope); NEEDS_HUMAN (a default-on behavior change or maintainer-owned schema choice); UNSAFE (a change would weaken a safety gate). +- Escalation: main owns the plan, FSM, git and delivery. Subagents are leaves; a packet two distinct agents fail is reclaimed by main (DISPATCH-RETIRE-01). +- Resource bounds: this checkout (`/Users/jun/.codex/worktrees/639c/codexclaw`), `gh` with the user's credentials, V1 subagents inheriting this session's model. Writes limited to the IN scope below. No token or wall-clock bound was stated; host limits apply. + +## Review and verification lanes + +At every phase's A, one read-only reviewer (`cxc-dev-code-reviewer`, `cxc-search`) audits the plan, and in parallel an inherited-model verifier checks the phase against the PABCD initiative rules at `/Users/jun/Developer/new/700_projects/pabcd_initiative/skills/dev-pabcd/SKILL.md` (DIFFLEVEL-ROADMAP-01, PHASE-SPLIT-01, LEXICO-SPLIT-01, UNIT-RESIDENCE-01, SOT-SYNC-01, C-ACTIVATION-GROUNDING-01, PLAN-VERIFIER-REAL-01, PLAN-FIELD-CHAIN-01, PLAN-BYPASS-NAMED-01, review rules). At every C, a fresh implementation reviewer and an initiative verifier run in parallel against the diff and fresh gate output. Verdicts are recorded in each decade doc's review section and in criterion c-9. + +## Scope and file map + +IN (details in each decade doc): + +``` +plugins/codexclaw/components/subagent-config/{src,dist,test}/dispatch-contract.* 010 (#276, #277) +plugins/codexclaw/skills/pabcd/references/delegation.md 010 +structure/INDEX.md (subagent-config section) 010 SoT sync +plugins/codexclaw/skills/interview/{SKILL.md,references/mind-dispatch.md} 020 (#275) +plugins/codexclaw/skills/loop/references/durable-goalplan.md 020, 030 +plugins/codexclaw/components/pabcd-state/{src,dist}/goalplan{,-cli}.*, test/goalplan-public-surface.test.ts 030 (#262) +plugins/codexclaw/skills/dev/references/async-questions.md 030 +README*.md badges, inventory.json, version files, CHANGELOG.md each phase (badges), 040 (release) +``` + +OUT: `plugins/codexclaw/hooks/*`, hook handlers and injected directive text (`pabcd-state/src/hook.ts`), `interview.ts`/`freeze*.ts` schema, Codex core/Desktop, anything deferred or declined in 001. + +## Ordered work phases + +Build order follows dependencies, not effort (PHASE-SPLIT-01). The three implementation phases touch disjoint files, so their order is set by delivery safety: the contract defect first, then guidance, then the goalplan schema extension, whose reviver change carries the most read-path risk and benefits from landing on a `dev` that already contains the other two. + +| Doc | Goalplan id | Phase | Depends on | +|---|---|---|---| +| 000-002 | wp1 | docs-only roadmap: this plan, triage, architect consultation | — | +| 010 | wp2 | dispatch contract: #276 verifier coverage, #277 verifier effects | wp1 | +| 020 | wp3 | Interview assumption provenance guidance (#275) | wp1 | +| 030 | wp5 | goalplan decision options (#262 follow-up) | wp1 | +| 040 | wp4 | delivery, issue disposition, main promotion, release 0.2.40 | wp2, wp3, wp5 | + +The goalplan id `wp5` was appended at this P as a LOOP-UNIT-CHAIN-01 amendment after triage found #262's options half small and self-contained; `wp4` gained `dependsOn: wp5` by a recorded hand edit (the CLI does not edit dependencies after creation). + +Delivery: one ordinary PR per implementation phase into `dev`, merged with a merge commit after hosted CI passes on its head, then the release PR, the `dev` -> `main` promotion and the release (040). No native stacks. + +## Issue acceptance mapping + +| Issue | Decision | Where | +|---|---|---| +| #276 | fix | 010 | +| #277 | implement | 010 | +| #275 | implement guidance | 020 | +| #262 | implement options; keep open for `withdrawn` | 030 | +| #213, #265 | close as completed | 001, 040 | +| #209, #247, #258, #259, #263, #264, #266, #267, #268 | close as not planned | 001, 040 | +| #255, #256, #257, #260, #273, #274 | keep open with a comment | 001, 040 | + +## SoT sync targets (SOT-SYNC-01) + +`structure/INDEX.md` (subagent-config file list, 010), `skills/interview/SKILL.md` (canonical owner of Interview rules, 020), `skills/loop/references/durable-goalplan.md` (goalplan schema and CLI, 020 and 030), `CHANGELOG.md` (040). + diff --git a/devlog/_plan/260930_issue_train/001_research.md b/devlog/_plan/260930_issue_train/001_research.md new file mode 100644 index 00000000..494a4b9b --- /dev/null +++ b/devlog/_plan/260930_issue_train/001_research.md @@ -0,0 +1,32 @@ +# Triage of the 20 open issues (2026-09-30) + +Three explorers read every open issue against `dev` at `659de59b` (read-only, source anchors in their returns). Five issues are new since the 2026-09-27 train (#273-#277). The other fifteen were re-verified against the reasons in `../260927_issue_train/001_research.md`. Only #255, #262 and #265 changed since then, each through a partial ship in that train. This train also applies a rule the user set on 2026-09-30: fix real defects and worthwhile improvements that need no maintainer decision, and do not grow the hook surface (no new hook files, no new hook injections or advisories). + +REAL means the shipped behavior is wrong. PROPOSAL means the report asks for new behavior. + +| Issue | Verdict | Grows hooks | Needs host signal | Decision | Reason | +|---|---|---|---|---|---| +| #276 receipt accepts an unrelated verifier result | REAL | no | no | fix (010) | `receiptSatisfiesPacket` never compares `verifierResult.command` with `packet.verifierCommands` and never checks coverage (`subagent-config/src/dispatch-contract.ts:124-129`). | +| #277 verifier write effects | PROPOSAL | no | no | implement (010) | Verifier commands are plain strings under one packet-level `worktreePolicy` (`dispatch-contract.ts:32,36`). An optional declaration plus a pure preflight answers the issue without executing anything. | +| #275 inferred vs confirmed assumptions at handoff | PROPOSAL | no | no | implement guidance (020) | The only closeout instruction is "Summarize the remaining OPEN ASSUMPTIONS" (`skills/interview/SKILL.md:172`). The prose plan section is hash-covered at freeze (`pabcd-state/src/freeze.ts:9-11`) and the Q/A ledger already gives answer ids (`interview-ledger.ts:29-42`), so guidance meets all four acceptance checks. The schema revision the issue mentions stays future work. | +| #262 decisions in goalplans | PARTIAL | no | no | implement `options[]` (030); keep open | #271 shipped open/decided decisions. `options[]` and "recommendation must be one of them" are specified in the issue's own acceptance list. `withdrawn` needs a maintainer decision because `openDecisionIdsForPhase` releases a phase only on `decided` (`goalplan.ts:1039-1043`). | +| #273 collection proof for independent workers | PROPOSAL | no | partial (unversioned `codex exec` rollout format) | keep open | Needs a receipt schema design first; the issue itself suggests a schema-and-fixtures PR. | +| #274 requirement provenance and revision trace | PROPOSAL | no | no | keep open | The issue asks the maintainer to decide goalplan vs ledger placement before implementation. | +| #255 SessionStart writes session state into every cwd | PARTIAL | no | no | keep open | `.gitignore` shipped (`codexclaw-dir.ts:4,19`); state is still created unconditionally (`hook.ts:604-607`). Lazy creation touches evidence and goal gates and needs its own design. | +| #256 independent verification receipt | PROPOSAL | no | no | keep open | Receipts store a joined command string with no argv, cwd or output digest (`receipt-cli.ts:171-179`); changing that changes C>D evidence. | +| #257 strict accepted-progress report | PROPOSAL | no | no | keep open | Depends on a post-implementation acceptance record (#256); review rounds are plan-audit only (`review-round-cli.ts:236`). | +| #260 unit fields and `cxc loop check` | PROPOSAL | no | no | keep open | Thresholds (chain length, criteria count) are maintainer policy; the external-wait part is now covered by `awaitsDecision`. | +| #213 automation ids are host-global | PARTIAL, plugin part done | no | yes | close as completed (plugin scope) | The ownership gate denies foreign, retargeted and ambiguous callers (`automation-ownership-gate.ts:44-60`) and the rule is documented (`skills/loop/references/waiting.md:116`). Atomic authorization inside the host mutation handler is Codex's job. | +| #265 on-disk worker progress checkpoint | PROPOSAL, guidance shipped | no | no | close as completed | The checkpoint convention shipped in #270 (`skills/pabcd/references/delegation.md:25-43`); a validation verb adds little over a three-field file. | +| #209 pending worktree thread id | NOT-REPRODUCED in plugin | no | yes | close as not planned | The gap is in the Desktop creation wrapper; the plugin already separates provisional and canonical ids (`dispatch-surfaces.md:134-160`, `scripts/check-lane-packet.mjs:34-50`). | +| #247 collab family at SessionStart | PROPOSAL | yes (new recorder or card injection) | yes | close as not planned | SessionStart carries no tool catalog (`fallback-dispatch-cli.ts:36-38`); the one-cell resolver already covers dispatch (`dispatch-card.ts:28-36`). | +| #258 verbatim prompt archive | PROPOSAL | yes (new UserPromptSubmit hook) | yes | close as not planned | The payload has no typed-versus-injected field (`parse.ts:71-77`), so the archive cannot promise verbatim user input. | +| #259 peer prompt provenance | PROPOSAL | yes (envelope parsing in UserPromptSubmit) | yes | close as not planned | An unauthenticated text envelope that changes hook behavior can be typed by anyone; the rule is already documented (`peer-collaboration.md:87-91`). | +| #263 PreCompact checkpoint hook | PROPOSAL | yes (new hook) | no | close as not planned | Adds a hook, which this train rules out. | +| #264 measured state after compaction | PROPOSAL | yes (new SessionStart injection) | no | close as not planned | New injection; listing background records also rewrites them (`bg-wake/src/registry.ts:110-158`). | +| #266 host hygiene doctor | PROPOSAL | yes (notice at SessionStart or phase boundary) | yes | close as not planned | Descriptor counts and exec-child ownership are host facts the plugin cannot establish. | +| #267 suggested wave width | PROPOSAL | yes (dispatch-card line) | yes | close as not planned | Family and open-child count are not observable at SessionStart; limits are documented (`dispatch-surfaces.md:195-201`). | +| #268 compact at a clean boundary | PROPOSAL | yes (Stop advisory) | yes | close as not planned | No token-usage signal is parsed, and Stop `systemMessage` display is unverified. | + +Issues kept open get one comment naming the reason and linking this file on `dev`. Closed issues get the reason line as a closing comment. + diff --git a/devlog/_plan/260930_issue_train/002_architect_consultation.md b/devlog/_plan/260930_issue_train/002_architect_consultation.md new file mode 100644 index 00000000..9a281c1b --- /dev/null +++ b/devlog/_plan/260930_issue_train/002_architect_consultation.md @@ -0,0 +1,49 @@ +# 002 — Architect consultation + +Architect: V1 subagent `01a0ee15-caad-7330-8b8d-4c028cefd432`, dispatched read-only with `cxc-dev` and `cxc-dev-architecture` attached (`CXC-ROLE: architect` header; the V1 schema has no `agent_type`). It inherited this session's model; no model override was requested. The proposal covered the three changes in `010`, `020` and `030` and returned 28 decisions. Main's dispositions follow; the reflection result is appended below after the architect checks the written plan. + +## Decisions and dispositions + +| ID | Decision (short) | Main | +|---|---|---| +| D1 | Shared `VerifierResult` type for legacy `verifierResult` and new `verifierResults[]` | accept (010 1a, 1c) | +| D2 | Merge both result fields, one rule set | accept (010 1g) | +| D3 | Distinct trimmed required commands; exact equality after trim | accept (010 1d `distinctCommands`) | +| D4 | Every required command needs a match; every result must exit 0 | accept | +| D5 | Unrelated results fail even with full coverage when commands are required; extras go in `commandsRun` | accept; the issue says an unrelated result must be rejected, and a gate fails closed. Disclosed in CHANGELOG and DISPATCH-VERIFIER-01 | +| D6 | Legacy single result with several required commands reports incomplete | accept | +| D7 | Additive `missing: string[]` in the return value | accept | +| D8 | `validateReceipt` checks both result fields' shape | accept | +| D9 | `validatePacket` rejects blank verifier command entries; duplicates allowed | accept | +| D10 | Optional `verifierEffects[]` beside unchanged `verifierCommands` | accept | +| D11 | Effect entries must name a listed command, once, with valid field types | accept | +| D12 | No path checks on `expectedWrites`; a declaration is a claim | accept | +| D13 | Preflight rules; declared writes on shared-read need isolation | accept | +| D14 | Preflight separate from validation; undeclared verifiers are flagged, not invalid | accept | +| D15 | DISPATCH-VERIFIER-01 paragraph in `delegation.md` after DISPATCH-TASK-01 | accept | +| D16 | Rebuild tracked dist files | accept | +| D17 | Inline provenance in the OPEN ASSUMPTIONS line; tracker text mirrors it | accept (020 1b) | +| D18 | Only proposed/open stay in OPEN ASSUMPTIONS and tracker; confirmed to requirements; rejected to a separate section | accept; section name `## ASSUMPTION DECISIONS` | +| D19 | Answer reference is the `answer_recorded` `eventId`, not a bare `questionId` | accept | +| D20 | `proposed` vs `open` meaning; high-impact proposed asked next; closeout groups | accept | +| D21 | `hook.ts:1935` unchanged; one clause in `durable-goalplan.md:40` | accept | +| D22 | `options?: string[]` on decisions | accept | +| D23 | Validation in `reviveDecisions` (fail closed) and the reviver copies `options` | accept; reviver keeps exact strings and compares trimmed, matching its existing rule, while `ask` stores trimmed values | +| D24 | `ask` mirrors the reviver rules with specific reasons; never stores `[]` | accept | +| D25 | Do not require the `decide` answer to be one of the options | accept. This overrides the goal objective's and criterion c-10's wording "answer validated against options when present": the host always offers a free-form reply, so the check would strand linked phases. c-10's evidence will state this disposition | +| D26 | Repeatable `--option`; no `options: []` default | accept | +| D27 | `ready --json` and `show` expose options; docs include `async-questions.md:56` | accept, including the async-questions sync | +| D28 | `withdrawn` out of scope; old plans round-trip | accept | + +## Unresolved assumptions and answers + +1. A legacy single result matching the only required command still passes: yes (existing test `:94` stays green). +2. Writes outside the checkout are not treated as safe under shared-read: yes (D13). +3. Minimum option count: one. The issue's acceptance list requires membership, not a count; two-or-more is a style choice left to the question author. +4. `async-questions.md:56` update: in scope. +5. Rejected-assumption section name: `## ASSUMPTION DECISIONS`. `listPlanFiles` (`freeze-cli.ts:29`) hashes whole files, so no heading set is required. +6. Goal-mode confirmation reference: a decided goalplan decision id (020 1b). +7. Trim-only matching is shared by validation and satisfaction: yes. + +## Reflection + diff --git a/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md b/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md new file mode 100644 index 00000000..e4e5e40c --- /dev/null +++ b/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md @@ -0,0 +1,294 @@ +# 010 — wp2: dispatch contract verifier matching (#276) and optional verifier effects (#277) + +A dispatch receipt now satisfies its packet only when every required verifier command has a matching successful result. An unrelated result is rejected, and an older receipt that reports one result for a packet with several commands reads as incomplete. Packets can also declare what each verifier writes, and a pure preflight tells the caller which verifiers need an isolated copy under a shared-read packet. Nothing executes a command, and no hook changes. + +## Phase contract + +- Work phase: `wp2` (goalplan), issues [#276](https://github.com/lidge-jun/codexclaw/issues/276) and [#277](https://github.com/lidge-jun/codexclaw/issues/277). Class C2: one module and its test file, no production consumer (`rg --no-ignore` finds only `test/dispatch-contract.test.ts:6-12`), public exported types change additively. +- Design decisions: D1-D16 in `002_architect_consultation.md`. +- Behavior change to disclose in CHANGELOG (040): a legacy single `verifierResult` whose command differs from the packet's only command, even cosmetically (`npm run test` vs `npm test`), no longer satisfies the packet; extra passing checks belong in `commandsRun`, not in `verifierResults`. + +## File change map + +Anchors checked against `codex/issue-train-0930` = `origin/dev` `659de59b` on 2026-09-30. + +### 1. MODIFY `plugins/codexclaw/components/subagent-config/src/dispatch-contract.ts` + +(a) After `DispatchStatus` (`:14`) add the two shared shapes (D1, D10): + +```ts +/** One verifier command's result as reported by the subagent (#276). */ +export interface VerifierResult { + command: string; + exitCode: number; + output: string; +} + +/** + * Declared write effects of one verifier command (#277). A declaration is the + * packet author's claim, not proof: codexclaw never runs the command and does + * not check paths against a filesystem. + */ +export interface VerifierEffect { + /** Must equal (after trim) one entry of `verifierCommands`. */ + command: string; + /** Paths or globs the command may write; `[]` declares it read-only. */ + expectedWrites: string[]; + /** Run this verifier in an isolated copy even when the packet is shared-read. */ + runInIsolation?: boolean; +} +``` + +(b) In `DispatchPacket`, after `verifierCommands` (`:31-32`): + +```diff + /** Verifier commands the main agent will run to check the result. */ + verifierCommands: string[]; ++ /** Optional write-effect declarations, at most one per verifier command (#277). */ ++ verifierEffects?: VerifierEffect[]; +``` + +(c) In `DispatchReceipt` (`:60-61`): + +```diff +- /** Verifier result from the subagent's perspective. */ +- verifierResult?: { command: string; exitCode: number; output: string }; ++ /** Legacy single verifier result; still read, merged with `verifierResults`. */ ++ verifierResult?: VerifierResult; ++ /** One result per packet verifier command (#276). Extra checks go in `commandsRun`. */ ++ verifierResults?: VerifierResult[]; +``` + +(d) Helpers, placed before `validatePacket` (`:66`): + +```ts +function isVerifierResult(value: unknown): value is VerifierResult { + if (!value || typeof value !== "object" || Array.isArray(value)) return false; + const v = value as Record; + return typeof v.command === "string" && Number.isInteger(v.exitCode) && typeof v.output === "string"; +} + +/** Distinct trimmed commands; exact string equality after trim, no other normalization (D3). */ +function distinctCommands(commands: readonly string[]): string[] { + return [...new Set(commands.map((command) => command.trim()))]; +} + +function validateVerifierEffects(value: unknown, commands: unknown): string[] { + if (!Array.isArray(value)) return ["verifierEffects must be an array"]; + const required = new Set(Array.isArray(commands) + ? commands.filter((command): command is string => typeof command === "string").map((command) => command.trim()) + : []); + const seen = new Set(); + const errors: string[] = []; + for (const item of value) { + if (!item || typeof item !== "object" || Array.isArray(item)) { + errors.push("verifierEffects entries must be objects"); + continue; + } + const effect = item as Record; + if (typeof effect.command !== "string" || !effect.command.trim()) { + errors.push("verifierEffects command must be a non-empty string"); + continue; + } + const command = effect.command.trim(); + if (!required.has(command)) errors.push("verifierEffects command `" + command + "` is not in verifierCommands"); + if (seen.has(command)) errors.push("verifierEffects declares `" + command + "` more than once"); + seen.add(command); + if (!Array.isArray(effect.expectedWrites) + || effect.expectedWrites.some((path) => typeof path !== "string" || !path.trim())) { + errors.push("verifierEffects expectedWrites for `" + command + "` must be an array of non-empty strings"); + } + if (effect.runInIsolation !== undefined && typeof effect.runInIsolation !== "boolean") { + errors.push("verifierEffects runInIsolation for `" + command + "` must be a boolean"); + } + } + return errors; +} +``` + +(e) `validatePacket`, after `:78` (D9, D11): + +```diff + if (!Array.isArray(p.verifierCommands)) errors.push("verifierCommands must be an array"); ++ else if (p.verifierCommands.some((command) => typeof command !== "string" || !command.trim())) { ++ errors.push("verifierCommands entries must be non-empty strings"); ++ } ++ if (p.verifierEffects !== undefined) errors.push(...validateVerifierEffects(p.verifierEffects, p.verifierCommands)); +``` + +(f) `validateReceipt`, after `:108` (D8): + +```diff + if (!Array.isArray(r.unresolvedAssumptions)) errors.push("unresolvedAssumptions must be an array"); ++ if (r.verifierResult !== undefined && !isVerifierResult(r.verifierResult)) { ++ errors.push("verifierResult must be {command: string, exitCode: integer, output: string}"); ++ } ++ if (r.verifierResults !== undefined ++ && (!Array.isArray(r.verifierResults) || !r.verifierResults.every(isVerifierResult))) { ++ errors.push("verifierResults must be an array of {command: string, exitCode: integer, output: string}"); ++ } +``` + +(g) Replace `receiptSatisfiesPacket` (`:112-131`) (D2-D7): + +```ts +/** + * Check that a receipt satisfies its packet's verifier requirements (#276). + * Every distinct required command needs a matching result, every result must + * exit 0, and a result for a command the packet did not require is rejected. + */ +export function receiptSatisfiesPacket(packet: DispatchPacket, receipt: DispatchReceipt): { + satisfied: boolean; + reasons: string[]; + /** Required verifier commands (trimmed) with no matching result. */ + missing: string[]; +} { + const reasons: string[] = []; + if (receipt.packetId !== packet.id) { + reasons.push("packetId mismatch: expected " + packet.id + " got " + receipt.packetId); + } + if (receipt.status !== "complete") { + reasons.push("receipt status is " + receipt.status + ", not complete"); + } + const required = distinctCommands(packet.verifierCommands); + const results = [ + ...(Array.isArray(receipt.verifierResults) ? receipt.verifierResults : []), + ...(receipt.verifierResult ? [receipt.verifierResult] : []), + ]; + if (required.length > 0 && results.length === 0) { + reasons.push("packet has verifier commands but receipt has no verifier result"); + } + for (const result of results) { + if (result.exitCode !== 0) { + reasons.push("verifier exit code " + result.exitCode + " (expected 0) for `" + result.command.trim() + "`"); + } + } + const reported = new Set(results.map((result) => result.command.trim())); + const missing = required.filter((command) => !reported.has(command)); + if (results.length > 0) { + for (const command of missing) reasons.push("missing verifier result for `" + command + "`"); + } + if (required.length > 1 && receipt.verifierResults === undefined && receipt.verifierResult) { + reasons.push("receipt reports one legacy verifierResult; packet requires " + + required.length + " verifier commands (incomplete)"); + } + if (required.length > 0) { + const requiredSet = new Set(required); + for (const command of reported) { + if (!requiredSet.has(command)) reasons.push("verifier result for unrelated command `" + command + "`"); + } + } + return { satisfied: reasons.length === 0, reasons, missing }; +} +``` + +(h) Append the preflight (D12-D14): + +```ts +/** One preflight row per distinct verifier command (#277). */ +export interface VerifierPreflightEntry { + command: string; + declared: boolean; + needsIsolation: boolean; + reason: string; +} + +/** + * Pure preflight over declared verifier effects (#277). It never runs a command + * or reads the filesystem; it only tells the caller which verifiers must not run + * on a shared checkout without isolation or main's confirmation. + */ +export function verifierPreflight(packet: DispatchPacket): VerifierPreflightEntry[] { + const effects = new Map((packet.verifierEffects ?? []).map((effect) => [effect.command.trim(), effect])); + return distinctCommands(packet.verifierCommands).map((command) => { + const effect = effects.get(command); + const declared = effect !== undefined; + if (packet.worktreePolicy === "isolated-write") { + return { command, declared, needsIsolation: false, reason: "packet is isolated-write" }; + } + if (!effect) { + return { command, declared, needsIsolation: true, + reason: "no declared write boundary on a shared-read packet; run it in an isolated copy or confirm with main" }; + } + if (effect.runInIsolation === true) { + return { command, declared, needsIsolation: true, reason: "declared runInIsolation" }; + } + if (effect.expectedWrites.length > 0) { + return { command, declared, needsIsolation: true, + reason: "declares writes (" + effect.expectedWrites.join(", ") + ") on a shared-read packet" }; + } + return { command, declared, needsIsolation: false, reason: "declared read-only (expectedWrites: [])" }; + }); +} +``` + +### 2. REGENERATE `plugins/codexclaw/components/subagent-config/dist/dispatch-contract.js` + +`npm run build` (`package.json` `"build": "node plugins/codexclaw/scripts/build.mjs"`). The file is tracked and `plugins/codexclaw/test/dist-freshness.test.mjs:27` compares it byte for byte with the compiled source. + +### 3. MODIFY `plugins/codexclaw/components/subagent-config/test/dispatch-contract.test.ts` + +Import `verifierPreflight` beside the existing imports (`:6-12`). Existing tests stay; `:94` additionally asserts `result.missing` deep-equals `[]`. New tests (each names the conditional path it drives): + +| Test name | Trigger | Observable effect | +|---|---|---| +| `receiptSatisfiesPacket: #276 repro — unrelated single result for two required commands` | packet `["first-check","second-check"]`, receipt `verifierResult: {command:"unrelated-check",exitCode:0}` | `satisfied:false`; reasons include `unrelated command`, `incomplete`; `missing` = both | +| `receiptSatisfiesPacket: single mismatched command is rejected` | packet `["npm test"]`, result `npm run test` | `satisfied:false`, `missing:["npm test"]` | +| `receiptSatisfiesPacket: two required commands with one matching verifierResults entry` | `verifierResults:[{first-check,0}]` | `satisfied:false`, `missing:["second-check"]`, reason `missing verifier result` | +| `receiptSatisfiesPacket: legacy single matching result with two required commands is incomplete` | `verifierResult:{first-check,0}`, no `verifierResults` | reason includes `legacy verifierResult` and `incomplete` | +| `receiptSatisfiesPacket: verifierResults covering every command is satisfied` | both commands exit 0 | `satisfied:true`, `reasons:[]`, `missing:[]` | +| `receiptSatisfiesPacket: unrelated extra result fails even with full coverage` | both required plus `extra-check` exit 0 | `satisfied:false`, reason naming `extra-check` as unrelated | +| `receiptSatisfiesPacket: duplicate and padded packet commands match trimmed results` | packet `[" npm test ","npm test"]`, result `npm test` | `satisfied:true` | +| `receiptSatisfiesPacket: nonzero matching result in verifierResults fails` | second-check exit 2 | reason includes `exit code 2` | +| `validateReceipt: rejects malformed verifier results` | `verifierResult.exitCode:"0"`; `verifierResults:{}`; `verifierResults:[{command:1}]` | each returns the matching error | +| `validatePacket: rejects blank verifier command entries` | `verifierCommands:["npm test"," "]` | error `entries must be non-empty strings` | +| `validatePacket: verifierEffects shape` | unknown command; duplicate command; `expectedWrites:"x"`; `runInIsolation:"yes"`; valid `[{command:"npm test",expectedWrites:[]}]` | four errors, then `[]` | +| `verifierPreflight: shared-read flags undeclared and writing verifiers` | shared-read with undeclared, `[]`, writes, `runInIsolation` | `needsIsolation` true/false/true/true, `declared` false/true/true/true | +| `verifierPreflight: isolated-write accepts declared writes` | isolated-write with writes | `needsIsolation:false`, `declared:true` | + +Thirteen new tests. The README test badges move by the measured delta (040). + +### 4. MODIFY `plugins/codexclaw/skills/pabcd/references/delegation.md` + +Insert after `:23` (`(DISPATCH-ECONOMY-01).`), before `### Optional worker progress checkpoint (#265)` (D15): + +```markdown + +**DISPATCH-VERIFIER-01 (DEFAULT).** When a packet names verifier commands, the +receipt reports one result per command; extra checks belong in the commands-run +list, not in the verifier results. A typed receipt satisfies its packet only when +every required command has a matching result with exit 0 and no result names a +command the packet did not require. Under a shared-read packet, declare each +verifier's writes (`expectedWrites: []` for read-only) or run it in an isolated +copy; an undeclared verifier goes back to main before it runs in a shared tree. +A declaration is the author's claim, not proof: codexclaw never executes it. +``` + +### 5. OPTIONAL SoT sync `structure/INDEX.md:138-140` + +The `components/subagent-config` section does not list `dispatch-contract.ts`; add one line: `- src/dispatch-contract.ts — typed DispatchPacket/DispatchReceipt (#17), verifier coverage (#276) and verifier effects preflight (#277)`. SOT-SYNC-01 target for this phase. + +## Scope boundary + +IN: the five files above. OUT: `commandsRun` semantics, `sourceIdentity`, release-gate's `dispatch-contracts` receipt (`pabcd-state/src/release-gate.ts:393`, stays missing), #256/#273 receipt fields, any hook. + +## Verification (PLAN-VERIFIER-REAL-01) + +| Command | Exit on `659de59b` | Reads the change target | +|---|---|---| +| `node plugins/codexclaw/scripts/test.mjs plugins/codexclaw/components/subagent-config/test/dispatch-contract.test.ts` | 0 (17 pass) | yes: the file is the direct argument and imports `../src/dispatch-contract.ts` | +| `npm run build` | 0 | yes: `build.mjs` recompiles every component `src` into `dist` | +| `node --test plugins/codexclaw/test/dist-freshness.test.mjs` | 0 | yes: compares tracked `dist/dispatch-contract.js` with the compiled source | +| `npm test` then `node plugins/codexclaw/scripts/inventory.mjs --check --tests ` | 0 | yes: root glob includes `subagent-config/test/*.test.ts`; inventory checks the README badge total | +| `node plugins/codexclaw/scripts/gate.mjs`, `node plugins/codexclaw/scripts/platform-smoke.mjs` | 0 | gate: inventory and skill checks; smoke: packaging. Neither observes the new logic; they guard regressions only | +| delegation.md prose | — | this command does not observe this change; human review in A and C | + +Red-green: the #276 repro test and the mismatched-command test must fail against the old `receiptSatisfiesPacket` (run them before replacing (g)), then pass. + +## Enforcement naming (PLAN-BYPASS-NAMED-01) + +- Tier: E2 (pure library check), executing surface: whichever caller invokes `receiptSatisfiesPacket`/`verifierPreflight`; no codexclaw runtime calls them today. +- Known bypass: a caller that never calls the functions, or a receipt author who puts an unrelated command's output under a required command's name. +- Residual risk: command strings are self-reported; the check proves coverage of names, not that the command ran. +- Wording: described as a contract check and preflight, never as enforcement. Final enforcement layer: none. + diff --git a/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md b/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md new file mode 100644 index 00000000..7460802c --- /dev/null +++ b/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md @@ -0,0 +1,128 @@ +# 020 — wp3: Interview assumption provenance (#275) + +Interview hands Plan two different things under one heading today: requirements the user agreed to and assumptions the assistant inferred. This phase adds one guidance rule, INTERVIEW-ASSUME-01, so each assumption carries its source, confidence, consequence if wrong and a status, and only unresolved ones travel as OPEN ASSUMPTIONS. Confirmed and rejected entries keep an answer reference from the existing Q/A ledger. No code, schema, hook or command changes. + +## Phase contract + +- Work phase: `wp3`, issue [#275](https://github.com/lidge-jun/codexclaw/issues/275). Class C1 per file, C2 as a set (three skill documents, guidance only). Design decisions D17-D21 in `002_architect_consultation.md`. +- Why guidance satisfies the four acceptance checks: no production CLI writes `tracker.assumptions` (the only writer, `pabcd-state/src/triage.ts:95-107`, has no caller outside tests), so the plan file's `## OPEN ASSUMPTIONS` section is the working record; the plan directory is hash-covered at freeze (`pabcd-state/src/freeze.ts:9-11`, `freeze-cli.ts:29,93`); freeze copies tracker text verbatim (`freeze-cli.ts:97`), so a status written into the line survives into the manifest; the answer ledger already mints `eventId` = `::answer_recorded` (`interview-ledger.ts:35-37`). +- Out of scope: the structured `Assumption` schema revision the issue mentions as future work (`interview.ts:48-57` and its reader at `:197-213` stay unchanged), `hook.ts:1935` runtime text (still correct: low/medium contradictions still become OPEN ASSUMPTIONS). + +## File change map + +Anchors checked against `659de59b`. + +### 1. MODIFY `plugins/codexclaw/skills/interview/SKILL.md` + +(a) `## Contract`, line 28: + +```diff +-- Record medium/low unresolved items as OPEN ASSUMPTIONS before leaving Interview. ++- Record medium/low unresolved items as OPEN ASSUMPTIONS before leaving Interview, ++ with the provenance fields of INTERVIEW-ASSUME-01. +``` + +(b) NEW section between `## Classify the loop before Plan` (ends `:44`) and `## Question quality (INTERVIEW-Q-01)` (`:46`): + +```markdown +## Assumption provenance (INTERVIEW-ASSUME-01) + +An assumption the assistant inferred is not a requirement the user agreed to, and +the handoff to Plan keeps the two apart. Write each assumption in the plan file as +one line: + + - A3 [proposed] Exports stay CSV only — source: src/export.ts:41; confidence: medium; if wrong: the XLSX writer and its tests join the scope + +- `source` is a repository `path:line`, or for something the user said, the + `eventId` of its `answer_recorded` event in the Q/A ledger + (`::answer_recorded`). A bare `questionId` is not enough; + it can repeat across turns. +- `confidence` is `low`, `medium` or `high`. `if wrong` names what changes in + scope, design or verification. +- Status is `proposed` (inferred, not yet asked), `open` (asked or deliberately + deferred, still unresolved), `user_confirmed` or `user_rejected`. The last two + require the answer's `eventId`; without one an entry stays `proposed` or + `open`, whatever the conversation seemed to imply. Under an active goal, where + Interview is suppressed, a decided goalplan decision id is the answer reference. +- Only `proposed` and `open` entries go under `## OPEN ASSUMPTIONS` and into the + tracker's assumptions: freeze carries every recorded tracker assumption into the + manifest as open. Move a `user_confirmed` entry into the plan's requirements with + its reference. Move a `user_rejected` entry under `## ASSUMPTION DECISIONS` with + its reference, so the decision stays traceable without being carried as open. +- This rule shapes existing plan text and tracker entries. It adds no field or + command, and older plans and trackers read as before. +``` + +(c) `## Question quality (INTERVIEW-Q-01)`, insert after the bullet ending `:53` (`so a larger independent batch has to be split across calls.`): + +```diff ++- High-impact `proposed` assumptions (INTERVIEW-ASSUME-01) are candidates for the ++ next relevant question round. Low-impact ones may stay `proposed`; closeout lists them. +``` + +(d) `## Rescan + readiness (INTERVIEW-SCAN-01)`, lines 170-172: + +```diff + - Treat readiness as a coverage claim on top of that: each dimension has concrete knowns, no + unresolved unknown changes scope, and every contradiction has exited into an answer or a +- recorded assumption. Summarize the remaining OPEN ASSUMPTIONS before claiming I -> P readiness. ++ recorded assumption. Before claiming I -> P readiness, summarize in two groups: confirmed ++ requirements with their answer references, then the remaining `proposed` and `open` ++ assumptions with their `if wrong` consequences (INTERVIEW-ASSUME-01). +``` + +(e) `## Closeout fork (INTERVIEW-FORK-01)`, lines 177-178: + +```diff +-`request_user_input` is hard-denied — see Goal firewall), after a scan round do not drift forward +-silently. Present a numbered choice and let the user pick: `1. Proceed to Plan` · ++`request_user_input` is hard-denied — see Goal firewall), after a scan round do not drift forward ++silently. Show the two-group summary from INTERVIEW-SCAN-01, then present a numbered choice and ++let the user pick: `1. Proceed to Plan` · +``` + +### 2. MODIFY `plugins/codexclaw/skills/interview/references/mind-dispatch.md`, lines 47-49 + +```diff + completed independent scan. Main triages high contradictions into questions and +-low/medium into OPEN ASSUMPTIONS, and records only actual authorized scan/tracker +-work. The existing answer-provenance/readiness and completion gates remain intact. ++low/medium into OPEN ASSUMPTIONS, which start as `proposed` under ++INTERVIEW-ASSUME-01, and records only actual authorized scan/tracker work. The ++existing answer-provenance/readiness and completion gates remain intact. +``` + +The file name stays; `minds.test.ts:38` checks that it exists. + +### 3. MODIFY `plugins/codexclaw/skills/loop/references/durable-goalplan.md`, line 40 + +```diff +-- Carry Interview OPEN ASSUMPTIONS into Plan/Audit instead of dropping them. ++- Carry Interview OPEN ASSUMPTIONS into Plan/Audit instead of dropping them, with ++ their source, confidence, consequence if wrong and status. Only `proposed` and ++ `open` entries count as open; confirmed and rejected entries keep their answer ++ reference (INTERVIEW-ASSUME-01 in cxc-interview). +``` + +SoT sync (SOT-SYNC-01): the Interview skill is the canonical owner of these rules (`pabcd/references/phase-plan.md:34`); `structure/INDEX.md:125,311` names the files only, so no structure doc changes. + +## Acceptance mapping (#275) + +| Check | Where it is met | +|---|---| +| 1. inference cannot be presented as user-confirmed without an answer reference | 1(b) status bullet: confirmed/rejected require the answer `eventId` | +| 2. rejected inference kept as decision trace, not carried as open | 1(b) last-but-one bullet: `## ASSUMPTION DECISIONS`, never in tracker assumptions | +| 3. closeout distinguishes confirmed requirements from open inferred assumptions | 1(d), 1(e) | +| 4. existing trackers and freeze manifests remain readable | no code change; 1(b) last bullet | + +## Verification (PLAN-VERIFIER-REAL-01) + +| Command | Exit on `659de59b` | Reads the change target | +|---|---|---| +| `rg -n 'INTERVIEW-ASSUME-01' plugins/codexclaw/skills` | 1 before (no match), expected 0 after with hits in the three files | yes: the three paths are under the searched directory | +| `node plugins/codexclaw/scripts/gate.mjs` | 0 | partly: skill frontmatter/inventory checks read `skills/interview/SKILL.md`; it does not read rule prose | +| `npm test` | 0 | no test reads these prose lines (explorer search of the rule strings found hits only in the SKILL.md and `hook.ts:1923,1932`); run as a regression guard | +| prose meaning | — | this command does not observe this change; human review in A (reviewer) and C (initiative verifier) | + +No conditional code path is added, so C-ACTIVATION-GROUNDING-01 does not apply; C-READER-01 applies to the new section (a fresh reader checks that the example line and the four statuses are understandable without this doc). + diff --git a/devlog/_plan/260930_issue_train/030_wp5_decision_options.md b/devlog/_plan/260930_issue_train/030_wp5_decision_options.md new file mode 100644 index 00000000..70b31570 --- /dev/null +++ b/devlog/_plan/260930_issue_train/030_wp5_decision_options.md @@ -0,0 +1,161 @@ +# 030 — wp5: goalplan decision options (#262 follow-up) + +A recorded goalplan decision can now carry the options that were offered, and when it does, its recommendation must be one of them. This closes the part of #262 that #271 left out: "`recommended` must be one of `options`; otherwise the verb refuses." Answers stay free text, because the host's question tool always adds a free-form choice. `withdrawn` remains out of scope, so #262 stays open. + +## Phase contract + +- Work phase: `wp5` (goalplan id; runs third, before delivery `wp4`), issue [#262](https://github.com/lidge-jun/codexclaw/issues/262). Class C2 with C4 care for the reviver (a malformed optional field must fail closed, as every existing field does). Design decisions D22-D28 in `002_architect_consultation.md`. +- No schema-version bump: `options` is optional and absent on old plans, which round-trip unchanged (`durable-goalplan.md:60`). +- Goalplan criterion c-10 records "answer validated against options when present". Architect D25 showed that rule would reject the host's free-form "Other" reply and strand linked phases, so this plan does not implement it; 002 records the disposition and c-10's evidence will state it. + +## Field chain (PLAN-FIELD-CHAIN-01) for `GoalplanDecision.options` + +| Stage | Path | +|---|---| +| Creation | CLI `--option` (`goalplan-cli.ts` parser) -> `runDecision` -> `askGoalplanDecision` input (`goalplan.ts:1225-1256`) | +| Serialization | `writeGoalplan` JSON.stringify of the plan object; no custom serializer (field order follows the object literal in `askGoalplanDecision`) | +| Deserialization | `reviveDecisions` (`goalplan.ts:526-550`), also reached by `invalidReason` (`:893`) | +| Consumers | `ready --json` open-decision projection (`goalplan-cli.ts:465-466`), `show` render (`:688-692`), `decideGoalplanDecision` (`goalplan.ts:1259-1275`: copies the decision with spread, so `options` survives; no membership check per D25), steering/other writers spread the plan and keep `decisions` untouched. Text `ready` (`:499-504`) prints id and question only; unchanged | + +## File change map + +### 1. MODIFY `plugins/codexclaw/components/pabcd-state/src/goalplan.ts` + +(a) Interface (`:141-149`): + +```diff + export interface GoalplanDecision { + id: string; + question: string; + recommendation?: string; ++ /** Options offered with the question; when present, recommendation is one of them. */ ++ options?: string[]; + status: "open" | "decided"; +``` + +(b) Before `reviveDecisions` (`:526`) add: + +```ts +/** Absent stays absent; a present list must be non-empty, non-blank and distinct after trim. */ +function reviveDecisionOptions(value: unknown): string[] | undefined | "invalid" { + if (value === undefined) return undefined; + if (!Array.isArray(value) || value.length === 0) return "invalid"; + if (value.some((option) => typeof option !== "string" || !option.trim())) return "invalid"; + const trimmed = (value as string[]).map((option) => option.trim()); + if (new Set(trimmed).size !== trimmed.length) return "invalid"; + return [...(value as string[])]; +} +``` + +(c) In `reviveDecisions`, after the shared field check (`:533-536`) and before `if (d.status === "open")`: + +```diff ++ const options = reviveDecisionOptions(d.options); ++ if (options === "invalid") return "invalid"; ++ if (options !== undefined && d.recommendation !== undefined ++ && !options.some((option) => option.trim() === (d.recommendation as string).trim())) return "invalid"; +``` + +and in both `decisions.push` calls (`:539-540`, `:543-545`) append `...(options === undefined ? {} : { options })` after the recommendation spread. Exact stored strings are kept (trim is only for comparison), matching the existing reviver's rule. + +(d) `askGoalplanDecision` (`:1225-1256`): + +```diff +- input: { id: string; question: string; recommendation?: string; workPhaseIds: string[]; askedAt: string }, ++ input: { id: string; question: string; recommendation?: string; options?: string[]; workPhaseIds: string[]; askedAt: string }, + ... + if (input.recommendation !== undefined && !recommendation) return { kind: "rejected", reason: "decision recommendation must not be empty" }; ++ const options = input.options?.map((option) => option.trim()); ++ if (options !== undefined) { ++ if (options.length === 0) return { kind: "rejected", reason: "decision options must not be empty" }; ++ if (options.some((option) => !option)) return { kind: "rejected", reason: "decision options must be non-empty text" }; ++ const repeated = options.find((option, index) => options.indexOf(option) !== index); ++ if (repeated !== undefined) return { kind: "rejected", reason: `duplicate decision option '${repeated}'` }; ++ if (recommendation !== undefined && !options.includes(recommendation)) { ++ return { kind: "rejected", reason: "decision recommendation must be one of the options" }; ++ } ++ } + ... + const decision: GoalplanDecision = { id, question, status: "open", askedAt: input.askedAt, +- ...(recommendation === undefined ? {} : { recommendation }) }; ++ ...(recommendation === undefined ? {} : { recommendation }), ++ ...(options === undefined ? {} : { options }) }; +``` + +### 2. MODIFY `plugins/codexclaw/components/pabcd-state/src/goalplan-cli.ts` + +(a) Args type near `:106-108`: add `options?: string[];` after `recommendation?: string;`. + +(b) `GoalplanFlag` (`:140-144`): add `| "--option"`. + +(c) `ask` rule (`:163`): add `"--option"` to `allowed` and `repeatable`; usage becomes `ask --session --id --question [--recommendation ] [--option ]... [--work-phase ]... [--cwd ]`. + +(d) Parser switch, after `case "--recommendation"` (`:239`); the default object at `:185` is NOT changed, so an `ask` without `--option` stores no `options` key (D26): + +```ts + case "--option": { + const option = value.trim(); + if (!option) return reject("--option requires one non-empty value"); + const options = out.options ?? (out.options = []); + if (options.includes(option)) return reject(`--option must not repeat '${option}'`); + options.push(option); + break; + } +``` + +(e) `runDecision` (`:530-533`): pass `...(args.options === undefined ? {} : { options: args.options })` into `askGoalplanDecision`. + +(f) `ready --json` (`:465-466`): project `options` additively: + +```diff +- .map(({ id, question, recommendation, askedAt }) => ({ id, question, ...(recommendation === undefined ? {} : { recommendation }), askedAt })); ++ .map(({ id, question, recommendation, options, askedAt }) => ({ id, question, ++ ...(recommendation === undefined ? {} : { recommendation }), ++ ...(options === undefined ? {} : { options }), askedAt })); +``` + +(g) `show` render (`:688-692`), after the question line: + +```diff + lines.push(` - ${decision.id} [open] ${decision.question}`); ++ if (decision.options !== undefined) { ++ lines.push(` options: ${decision.options.join(" | ")}${decision.recommendation === undefined ? "" : ` (recommended: ${decision.recommendation})`}`); ++ } +``` + +### 3. REGENERATE `pabcd-state/dist/goalplan.js`, `pabcd-state/dist/goalplan-cli.js` (`npm run build`; tracked, checked by `dist-freshness.test.mjs`). + +### 4. MODIFY `plugins/codexclaw/components/pabcd-state/test/goalplan-public-surface.test.ts` + +Beside the existing decision tests (`:684-764`), using the same temp-dir CLI helpers: + +| Test | Trigger | Observable effect | +|---|---|---| +| `ask records options and ready/show expose them` | `ask --option A --option B --recommendation A` | goalplan decision has `options:["A","B"]`; `ready --json` open decision has `options`; `show` prints `options: A \| B (recommended: A)` | +| `ask rejects a recommendation outside the options without a write` | `--option A --option B --recommendation C` | exit 1, reason `must be one of the options`, plan bytes unchanged | +| `ask rejects blank and repeated options at parse time` | `--option " "`; `--option A --option A` | exit 1 with the parser reasons, no write | +| `ask without --option stores no options key` | plain ask | decision JSON has no `options` property | +| `decide keeps options and accepts a free-form answer` | ask with options, `decide --answer "something else"` | exit 0, decision decided, `options` unchanged | +| `reviver fails closed on malformed options` | hand-written plans with `options: []`, `["A","A "]`, `[1]`, and a recommendation outside valid options | `cxc loop validate` / read reports the `decisions` field invalid | + +### 5. MODIFY docs + +- `plugins/codexclaw/skills/loop/references/durable-goalplan.md:60`: `{ id, question, recommendation?, options?, status: open|decided, answer?, askedAt, decidedAt? }` and one sentence: "When `options` is present it is a non-empty list of distinct entries and `recommendation` must be one of them; the answer stays free text because the host always offers a free-form reply." +- `durable-goalplan.md:98`: add `[--option ]...` to the `ask` synopsis. +- `plugins/codexclaw/skills/dev/references/async-questions.md:56`: add `[--option ]...` to the `ask` synopsis and "(the offered options, recommended first)". + +## Verification (PLAN-VERIFIER-REAL-01) + +| Command | Exit on `659de59b` | Reads the change target | +|---|---|---| +| `node plugins/codexclaw/scripts/test.mjs plugins/codexclaw/components/pabcd-state/test/goalplan-public-surface.test.ts` | 0 | yes: drives the CLI and imports `../src/goalplan*.ts` | +| `node plugins/codexclaw/scripts/test.mjs plugins/codexclaw/components/pabcd-state/test/goalplan.test.ts plugins/codexclaw/components/pabcd-state/test/goalplan-integrity.test.ts plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts` | 0 | yes: round-trip, integrity and Stop decision-wait paths read the reviver | +| `npm run build`, `dist-freshness.test.mjs`, `npm test`, `inventory.mjs --check`, `gate.mjs`, `platform-smoke.mjs` | 0 | as in 010 | +| docs prose | — | not observed by a command; human review | + +Activation scenarios (C-ACTIVATION-GROUNDING-01): each rejection branch in (1c), (1d) and (2d) is driven by a named test above; the free-form `decide` path proves D25. + +## Enforcement naming (PLAN-BYPASS-NAMED-01) + +Tier E2 (CLI and reviver validation). Executing surface: `cxc loop ask` and every goalplan read. Known bypass: none for stored plans (hand edits fail closed on read); the recommendation check does not prove the question was actually sent with those options. Residual risk: an invalid hand edit makes the whole plan unreadable until repaired, as for every existing field. Wording: validation, not enforcement of what the host displayed. + diff --git a/devlog/_plan/260930_issue_train/040_wp4_delivery.md b/devlog/_plan/260930_issue_train/040_wp4_delivery.md new file mode 100644 index 00000000..9181ff34 --- /dev/null +++ b/devlog/_plan/260930_issue_train/040_wp4_delivery.md @@ -0,0 +1,41 @@ +# 040 — wp4: delivery, issue disposition, main promotion and release 0.2.40 + +This phase lands nothing new in product code. It gets each implementation phase into `dev` through an ordinary PR with hosted CI read on the exact head, dispositions every open issue per `001_research.md`, bumps the version to 0.2.40, promotes `dev` to `main`, and publishes v0.2.40 through `release.yml` with verified assets. It reuses the 0.2.39 train (`../260924_issue_sweep_0238/070_release_0239.md`, `071_delivery.md`) with fresh evidence at every step. + +## Per implementation phase (wp2, wp3, wp5) + +1. Branch: wp2 is `codex/issue-train-0930-wp2`, cut from this session's `codex/issue-train-0930` (= `origin/dev` `659de59b` plus this unit's docs). wp3 and wp5 branch from the latest `origin/dev` after the previous PR merges; if the previous PR is still in CI, branch from its phase branch and retarget to `dev` after it merges (the 0927 train's amendment). +2. Local gates at the phase C: `npm run build`; focused tests through `cxc receipt test`; `npm test` (record the TAP total); `node plugins/codexclaw/scripts/inventory.mjs --write --tests ` then `--check --tests `; `node plugins/codexclaw/scripts/gate.mjs`; `node plugins/codexclaw/scripts/platform-smoke.mjs`; `git diff --stat origin/dev -- plugins/codexclaw/hooks plugins/codexclaw/.codex-plugin` shows no hook registration change (criterion c-5). +3. Privacy self-check before the first push (DEV-PRIVACY-01): grep the push range for tokens, client names and home paths other than this user's own (`git log -p origin/dev..HEAD | rg -i 'ghp_|sk-|token=|/Users/(?!jun)'`). +4. `git push -u origin `; `gh pr create --base dev --body-file ` (problem, behavior before/after, tests, residual risk, `Fixes #276` / `Fixes #277` / `Fixes #275` for fully fixed issues; #262 is referenced without `Fixes`). +5. Hosted CI (DEV-CI-EVIDENCE-01): `gh pr view --json headRefOid,statusCheckRollup`, `gh run list --commit --json databaseId,event,headSha,status,conclusion,workflowName`; confirm the ci.yml jobs (ubuntu, macOS, Windows shards, packed install, artifacts), labeler and target check ran on that head, distinguishing pending, skipped, cancelled and failed. Diagnose a failure from its job log before any rerun. +6. Merge with a merge commit (same method as #269-#272) using `gh pr merge --merge --match-head-commit `. + +## Issue disposition (after the implementation PRs merge) + +- #275, #276, #277 close through `Fixes`; verify each shows closed with the PR link. +- #262: comment naming the wp5 PR (`options[]` and recommendation membership shipped; answers stay free text by design; `withdrawn` still needs a maintainer decision); stays open. +- Close as not planned with the one-line reason from 001 and a link to `devlog/_plan/260930_issue_train/001_research.md` on `dev`: #209, #247, #258, #259, #263, #264, #266, #267, #268. +- Close as completed with the reason and evidence anchors: #213 (plugin scope), #265. +- Comment and keep open: #255, #256, #257, #260, #273, #274. + +## Release 0.2.40 + +1. Branch `codex/release-0240` from `origin/dev` after the implementation PRs merge. +2. Version files, found with `rg --hidden -l '0\.2\.39' --glob '!.git' --glob '!CHANGELOG.md' --glob '!devlog/**' --glob '!node_modules/**'`: `package.json`, `package-lock.json` (every `"version": "0.2.39"` owned by this repo's workspaces), `cli/package.json`, `plugins/codexclaw/gui/package.json`, `plugins/codexclaw/components/*/package.json` -> 0.2.40; `plugins/codexclaw/.codex-plugin/plugin.json` -> `0.2.40+codex.`; `inventory.json` and README badges regenerated with `inventory.mjs --write --tests `. `pabcd-state/test/hook.test.ts` matches the pattern: inspect it and leave it unless it asserts the current package version. +3. CHANGELOG: rename `[Unreleased]` to `[0.2.40] - ` keeping the 0927 train entries, and add: Added — optional `verifierEffects` and `verifierPreflight` (#277), `verifierResults` on receipts (#276), goalplan decision `options` / `cxc loop ask --option` (#262), Interview assumption provenance guidance INTERVIEW-ASSUME-01 (#275). Fixed — `receiptSatisfiesPacket` requires a matching successful result per required verifier command (#276). Compatibility — a legacy single `verifierResult` for a different command, or one result for a multi-command packet, no longer satisfies; two hooks from the 0927 train need trust approval after upgrade. +4. Local gates as above plus `node plugins/codexclaw/scripts/check-versions.mjs 0.2.40` and `npm ls --workspaces --depth=0`; PR to `dev`, exact-head CI, merge; dev push CI, Packed install and WSL green at the merge SHA. +5. Promotion PR `dev` -> `main` titled "Release codexclaw 0.2.40"; exact-head CI; merge with a merge commit; main push CI and Packed install green at the merge SHA. +6. `gh workflow run release.yml -f version=0.2.40 -f prerelease=false -f dry_run=true -f expected_sha=

`; read `release verify: READY — 0.2.40` and the test totals. Then the same with `dry_run=false`. +7. Assets: `gh release download v0.2.40`; `shasum -a 256 -c SHA256SUMS`; unpack the payload and compare with `git archive
plugins/codexclaw` (0 differing, missing or extra files); tag `v0.2.40` points at the main merge SHA; release is stable and latest. + +The installed plugin cache and remote hosts are not updated by this train (goal scope OUT). + +## Delivery record + +`041_delivery.md`: PRs, heads, merge commits, CI runs, release runs, asset verification, issue dispositions, residual risks. It lands on `dev` through a small docs PR, which then reaches `main` with the next promotion; that later promotion is outside this train. + +## Acceptance + +All goalplan criteria met with captured evidence; `cxc loop validate` passes; v0.2.40 is the latest release and its assets verify. + From 048a521c84bb8807f407ff8b966ef54784074199 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Wed, 30 Sep 2026 02:03:43 +0900 Subject: [PATCH 72/90] docs(plan): fold architect reflection --- devlog/_plan/260930_issue_train/002_architect_consultation.md | 4 ++++ .../_plan/260930_issue_train/020_wp3_interview_assumptions.md | 2 ++ devlog/_plan/260930_issue_train/030_wp5_decision_options.md | 2 +- 3 files changed, 7 insertions(+), 1 deletion(-) diff --git a/devlog/_plan/260930_issue_train/002_architect_consultation.md b/devlog/_plan/260930_issue_train/002_architect_consultation.md index 9a281c1b..532e0109 100644 --- a/devlog/_plan/260930_issue_train/002_architect_consultation.md +++ b/devlog/_plan/260930_issue_train/002_architect_consultation.md @@ -47,3 +47,7 @@ Architect: V1 subagent `01a0ee15-caad-7330-8b8d-4c028cefd432`, dispatched read-o ## Reflection + +Round 1 (same architect, after the plan was written): `REFLECTION: MISALIGNED` with one gap. D17's second half, that a tracker assumption's `text` must repeat the provenance line because freeze copies tracker text into the manifest (`freeze-cli.ts:97`), was missing from the executable guidance in 020. It also noted that 030's malformed-options test must assert the reviver's invalid-plan message naming `decisions` rather than a validate line, and confirmed that the code in 010 and 030 survives the type-stripping build and keeps existing tests green. + +Main disposition: both folded. 020 1(b) gained the tracker-text bullet; 030's reviver test row now asserts the `decisions` message. The optional `renderGoalplanHelp` wording is left to B. With the one gap closed, every decision D1-D28 maps to a plan location in the architect's table; main records the consultation as aligned after the fold. diff --git a/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md b/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md index 7460802c..78bd2baa 100644 --- a/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md +++ b/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md @@ -49,6 +49,8 @@ one line: manifest as open. Move a `user_confirmed` entry into the plan's requirements with its reference. Move a `user_rejected` entry under `## ASSUMPTION DECISIONS` with its reference, so the decision stays traceable without being carried as open. +- A tracker assumption's `text` repeats the same line without its leading `- ` + (freeze prepends it), so the frozen manifest keeps the provenance. - This rule shapes existing plan text and tracker entries. It adds no field or command, and older plans and trackers read as before. ``` diff --git a/devlog/_plan/260930_issue_train/030_wp5_decision_options.md b/devlog/_plan/260930_issue_train/030_wp5_decision_options.md index 70b31570..2072d80a 100644 --- a/devlog/_plan/260930_issue_train/030_wp5_decision_options.md +++ b/devlog/_plan/260930_issue_train/030_wp5_decision_options.md @@ -136,7 +136,7 @@ Beside the existing decision tests (`:684-764`), using the same temp-dir CLI hel | `ask rejects blank and repeated options at parse time` | `--option " "`; `--option A --option A` | exit 1 with the parser reasons, no write | | `ask without --option stores no options key` | plain ask | decision JSON has no `options` property | | `decide keeps options and accepts a free-form answer` | ask with options, `decide --answer "something else"` | exit 0, decision decided, `options` unchanged | -| `reviver fails closed on malformed options` | hand-written plans with `options: []`, `["A","A "]`, `[1]`, and a recommendation outside valid options | `cxc loop validate` / read reports the `decisions` field invalid | +| `reviver fails closed on malformed options` | hand-written plans with `options: []`, `["A","A "]`, `[1]`, and a recommendation outside valid options | the plan reads as invalid naming the `decisions` field (reviver path via `invalidReason`, `goalplan.ts:893`) before `validateGoalplan` runs; the test asserts that message, exact text confirmed at B | ### 5. MODIFY docs From dfdb4b015214cbed5eca66c44f1b38aee0f5a221 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Wed, 30 Sep 2026 02:14:31 +0900 Subject: [PATCH 73/90] docs(plan): fold A round 1 (activation tests, field chain, bypass, delivery fixes) --- devlog/_plan/260930_issue_train/000_plan.md | 9 +++---- .../_plan/260930_issue_train/001_research.md | 5 ++-- .../002_architect_consultation.md | 6 ++++- .../010_wp2_dispatch_contract.md | 26 ++++++++++++++----- .../020_wp3_interview_assumptions.md | 20 +++++++++----- .../030_wp5_decision_options.md | 11 ++++---- .../260930_issue_train/040_wp4_delivery.md | 13 +++++----- 7 files changed, 56 insertions(+), 34 deletions(-) diff --git a/devlog/_plan/260930_issue_train/000_plan.md b/devlog/_plan/260930_issue_train/000_plan.md index 5a37c9cb..4c1b67a8 100644 --- a/devlog/_plan/260930_issue_train/000_plan.md +++ b/devlog/_plan/260930_issue_train/000_plan.md @@ -15,7 +15,8 @@ Reader: a maintainer deciding whether to merge these changes and publish 0.2.40; - Memory artifact: this unit, the goalplan `.codexclaw/goalplans/codexclaw-issue-train-2026-09-30-repo-lidge-jun/`, and `.codexclaw/evidence/01a0ee0f-ea9e-7273-91fe-0c4188c2cae6/`. - Expected terminal outcomes: DONE (merged, released, dispositioned); BLOCKED (CI infrastructure, branch protection or credentials outside scope); NEEDS_HUMAN (a default-on behavior change or maintainer-owned schema choice); UNSAFE (a change would weaken a safety gate). - Escalation: main owns the plan, FSM, git and delivery. Subagents are leaves; a packet two distinct agents fail is reclaimed by main (DISPATCH-RETIRE-01). -- Resource bounds: this checkout (`/Users/jun/.codex/worktrees/639c/codexclaw`), `gh` with the user's credentials, V1 subagents inheriting this session's model. Writes limited to the IN scope below. No token or wall-clock bound was stated; host limits apply. +- Resource bounds: this checkout (`/Users/jun/.codex/worktrees/639c/codexclaw`), `gh` with the user's credentials, V1 subagents inheriting this session's model. Writes limited to the IN scope below. No token or wall-clock bound was stated; host limits apply. The initiative's loop-engineering rule asks for a token and wall-clock bound on C4 work; the release (040) is C4, and the user authorized it explicitly without a bound, so this is a disclosed gap rather than an invented budget. +- Review independence: every subagent inherits this session's model, as the user asked, so REVIEW-DECORRELATE-01's different-family review is not established; independence here is separate context only. ## Review and verification lanes @@ -40,7 +41,7 @@ OUT: `plugins/codexclaw/hooks/*`, hook handlers and injected directive text (`pa ## Ordered work phases -Build order follows dependencies, not effort (PHASE-SPLIT-01). The three implementation phases touch disjoint files, so their order is set by delivery safety: the contract defect first, then guidance, then the goalplan schema extension, whose reviver change carries the most read-path risk and benefits from landing on a `dev` that already contains the other two. +Build order follows dependencies, not effort (PHASE-SPLIT-01). The three implementation phases touch disjoint code; the one shared file is `durable-goalplan.md`, edited by 020 (line 40) and 030 (lines 60 and 98), so 030 runs after 020 and re-anchors its doc edits. Their order is set by delivery safety: the contract defect first, then guidance, then the goalplan schema extension, whose reviver change carries the most read-path risk and benefits from landing on a `dev` that already contains the other two. | Doc | Goalplan id | Phase | Depends on | |---|---|---|---| @@ -62,11 +63,9 @@ Delivery: one ordinary PR per implementation phase into `dev`, merged with a mer | #277 | implement | 010 | | #275 | implement guidance | 020 | | #262 | implement options; keep open for `withdrawn` | 030 | -| #213, #265 | close as completed | 001, 040 | -| #209, #247, #258, #259, #263, #264, #266, #267, #268 | close as not planned | 001, 040 | +| #209, #213, #247, #258, #259, #263, #264, #265, #266, #267, #268 | close as not planned | 001, 040 | | #255, #256, #257, #260, #273, #274 | keep open with a comment | 001, 040 | ## SoT sync targets (SOT-SYNC-01) `structure/INDEX.md` (subagent-config file list, 010), `skills/interview/SKILL.md` (canonical owner of Interview rules, 020), `skills/loop/references/durable-goalplan.md` (goalplan schema and CLI, 020 and 030), `CHANGELOG.md` (040). - diff --git a/devlog/_plan/260930_issue_train/001_research.md b/devlog/_plan/260930_issue_train/001_research.md index 494a4b9b..ecabade3 100644 --- a/devlog/_plan/260930_issue_train/001_research.md +++ b/devlog/_plan/260930_issue_train/001_research.md @@ -16,8 +16,8 @@ REAL means the shipped behavior is wrong. PROPOSAL means the report asks for new | #256 independent verification receipt | PROPOSAL | no | no | keep open | Receipts store a joined command string with no argv, cwd or output digest (`receipt-cli.ts:171-179`); changing that changes C>D evidence. | | #257 strict accepted-progress report | PROPOSAL | no | no | keep open | Depends on a post-implementation acceptance record (#256); review rounds are plan-audit only (`review-round-cli.ts:236`). | | #260 unit fields and `cxc loop check` | PROPOSAL | no | no | keep open | Thresholds (chain length, criteria count) are maintainer policy; the external-wait part is now covered by `awaitsDecision`. | -| #213 automation ids are host-global | PARTIAL, plugin part done | no | yes | close as completed (plugin scope) | The ownership gate denies foreign, retargeted and ambiguous callers (`automation-ownership-gate.ts:44-60`) and the rule is documented (`skills/loop/references/waiting.md:116`). Atomic authorization inside the host mutation handler is Codex's job. | -| #265 on-disk worker progress checkpoint | PROPOSAL, guidance shipped | no | no | close as completed | The checkpoint convention shipped in #270 (`skills/pabcd/references/delegation.md:25-43`); a validation verb adds little over a three-field file. | +| #213 automation ids are host-global | PARTIAL, plugin part done | no | yes | close as not planned (remainder is host scope) | The ownership gate denies foreign, retargeted and ambiguous callers (`automation-ownership-gate.ts:44-60`) and the rule is documented (`skills/loop/references/waiting.md:116`). Atomic authorization inside the host mutation handler is Codex's job. | +| #265 on-disk worker progress checkpoint | PROPOSAL, guidance shipped | no | no | close as not planned (validator declined) | The checkpoint convention shipped in #270 (`skills/pabcd/references/delegation.md:25-43`); the issue's remaining acceptance tests are for a validation verb, which adds little over a three-field file. | | #209 pending worktree thread id | NOT-REPRODUCED in plugin | no | yes | close as not planned | The gap is in the Desktop creation wrapper; the plugin already separates provisional and canonical ids (`dispatch-surfaces.md:134-160`, `scripts/check-lane-packet.mjs:34-50`). | | #247 collab family at SessionStart | PROPOSAL | yes (new recorder or card injection) | yes | close as not planned | SessionStart carries no tool catalog (`fallback-dispatch-cli.ts:36-38`); the one-cell resolver already covers dispatch (`dispatch-card.ts:28-36`). | | #258 verbatim prompt archive | PROPOSAL | yes (new UserPromptSubmit hook) | yes | close as not planned | The payload has no typed-versus-injected field (`parse.ts:71-77`), so the archive cannot promise verbatim user input. | @@ -29,4 +29,3 @@ REAL means the shipped behavior is wrong. PROPOSAL means the report asks for new | #268 compact at a clean boundary | PROPOSAL | yes (Stop advisory) | yes | close as not planned | No token-usage signal is parsed, and Stop `systemMessage` display is unverified. | Issues kept open get one comment naming the reason and linking this file on `dev`. Closed issues get the reason line as a closing comment. - diff --git a/devlog/_plan/260930_issue_train/002_architect_consultation.md b/devlog/_plan/260930_issue_train/002_architect_consultation.md index 532e0109..5eff9619 100644 --- a/devlog/_plan/260930_issue_train/002_architect_consultation.md +++ b/devlog/_plan/260930_issue_train/002_architect_consultation.md @@ -50,4 +50,8 @@ Architect: V1 subagent `01a0ee15-caad-7330-8b8d-4c028cefd432`, dispatched read-o Round 1 (same architect, after the plan was written): `REFLECTION: MISALIGNED` with one gap. D17's second half, that a tracker assumption's `text` must repeat the provenance line because freeze copies tracker text into the manifest (`freeze-cli.ts:97`), was missing from the executable guidance in 020. It also noted that 030's malformed-options test must assert the reviver's invalid-plan message naming `decisions` rather than a validate line, and confirmed that the code in 010 and 030 survives the type-stripping build and keeps existing tests green. -Main disposition: both folded. 020 1(b) gained the tracker-text bullet; 030's reviver test row now asserts the `decisions` message. The optional `renderGoalplanHelp` wording is left to B. With the one gap closed, every decision D1-D28 maps to a plan location in the architect's table; main records the consultation as aligned after the fold. +Main disposition: both folded. 020 1(b) gained the tracker-text bullet; 030's reviver test row now asserts the `decisions` message. The optional `renderGoalplanHelp` wording is left to B. The architect's verdict stays MISALIGNED with one gap, which main folded; the architect's table maps every other decision to a plan location. + +## Criterion c-10 correction (A round 1) + +Criterion c-10 was registered at this P, before D25 arrived, with the words "answer validated against options when present". D25 showed that rule would reject the host's free-form reply and strand linked phases; neither the user's request nor issue #262 asks for answer membership (the issue asks only that the recommendation be one of the options). Both A auditors flagged that meeting c-10 with an explanatory note would certify text that is not true. Main therefore corrected c-10's scenario to the delivered behavior through a recorded steering `annotate` entry with this rationale, before any implementation. This is a correction of main's own drafting error at P, not a change to a user requirement; it is disclosed in the D summary and the final report. diff --git a/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md b/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md index e4e5e40c..b843e3d2 100644 --- a/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md +++ b/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md @@ -8,6 +8,14 @@ A dispatch receipt now satisfies its packet only when every required verifier co - Design decisions: D1-D16 in `002_architect_consultation.md`. - Behavior change to disclose in CHANGELOG (040): a legacy single `verifierResult` whose command differs from the packet's only command, even cosmetically (`npm run test` vs `npm test`), no longer satisfies the packet; extra passing checks belong in `commandsRun`, not in `verifierResults`. +## Field chain (PLAN-FIELD-CHAIN-01) + +| Field | Creation | Serialization | Deserialization | Consumers | +|---|---|---|---|---| +| `DispatchReceipt.verifierResults` | N/A: receipts are JSON written by the dispatched agent or its caller; codexclaw has no builder or CLI for them (`rg --no-ignore` finds only the test) | N/A: plain JSON, no custom serializer | `validateReceipt` shape check (1f) | `receiptSatisfiesPacket` (1g); tests | +| `DispatchPacket.verifierEffects` | N/A: packets are hand-authored JSON; no builder | N/A: plain JSON | `validatePacket` (1e, via `validateVerifierEffects`) | `verifierPreflight` (1h); tests | +| return field `missing` | `receiptSatisfiesPacket` (1g) | N/A: in-memory return value | N/A | callers of `receiptSatisfiesPacket`; only the test today | + ## File change map Anchors checked against `codex/issue-train-0930` = `origin/dev` `659de59b` on 2026-09-30. @@ -151,10 +159,15 @@ export function receiptSatisfiesPacket(packet: DispatchPacket, receipt: Dispatch reasons.push("receipt status is " + receipt.status + ", not complete"); } const required = distinctCommands(packet.verifierCommands); - const results = [ + const reportedResults: unknown[] = [ ...(Array.isArray(receipt.verifierResults) ? receipt.verifierResults : []), ...(receipt.verifierResult ? [receipt.verifierResult] : []), ]; + // Unvalidated input must not throw here; malformed entries fail the receipt. + const results = reportedResults.filter(isVerifierResult); + if (results.length !== reportedResults.length) { + reasons.push("receipt has malformed verifier results (see validateReceipt)"); + } if (required.length > 0 && results.length === 0) { reasons.push("packet has verifier commands but receipt has no verifier result"); } @@ -224,7 +237,7 @@ export function verifierPreflight(packet: DispatchPacket): VerifierPreflightEntr ### 2. REGENERATE `plugins/codexclaw/components/subagent-config/dist/dispatch-contract.js` -`npm run build` (`package.json` `"build": "node plugins/codexclaw/scripts/build.mjs"`). The file is tracked and `plugins/codexclaw/test/dist-freshness.test.mjs:27` compares it byte for byte with the compiled source. +`npm run build` (`package.json` `"build": "node plugins/codexclaw/scripts/build.mjs"`). The file is tracked and `plugins/codexclaw/test/dist-freshness.test.mjs:28` compares it byte for byte with the compiled source. ### 3. MODIFY `plugins/codexclaw/components/subagent-config/test/dispatch-contract.test.ts` @@ -243,10 +256,12 @@ Import `verifierPreflight` beside the existing imports (`:6-12`). Existing tests | `validateReceipt: rejects malformed verifier results` | `verifierResult.exitCode:"0"`; `verifierResults:{}`; `verifierResults:[{command:1}]` | each returns the matching error | | `validatePacket: rejects blank verifier command entries` | `verifierCommands:["npm test"," "]` | error `entries must be non-empty strings` | | `validatePacket: verifierEffects shape` | unknown command; duplicate command; `expectedWrites:"x"`; `runInIsolation:"yes"`; valid `[{command:"npm test",expectedWrites:[]}]` | four errors, then `[]` | +| `validatePacket: verifierEffects container and entry guards` | `verifierEffects:{}`; `verifierEffects:[null]`; `verifierEffects:[{command:" ",expectedWrites:[]}]` | errors `must be an array`, `entries must be objects`, `command must be a non-empty string` | +| `receiptSatisfiesPacket: malformed unvalidated result does not throw` | `verifierResults:[{command:1}]` cast past the type | returns `satisfied:false` with the `malformed verifier results` reason | | `verifierPreflight: shared-read flags undeclared and writing verifiers` | shared-read with undeclared, `[]`, writes, `runInIsolation` | `needsIsolation` true/false/true/true, `declared` false/true/true/true | | `verifierPreflight: isolated-write accepts declared writes` | isolated-write with writes | `needsIsolation:false`, `declared:true` | -Thirteen new tests. The README test badges move by the measured delta (040). +Fifteen new tests. The README test badges move by the measured delta (040). ### 4. MODIFY `plugins/codexclaw/skills/pabcd/references/delegation.md` @@ -264,9 +279,9 @@ copy; an undeclared verifier goes back to main before it runs in a shared tree. A declaration is the author's claim, not proof: codexclaw never executes it. ``` -### 5. OPTIONAL SoT sync `structure/INDEX.md:138-140` +### 5. SoT sync `structure/INDEX.md:138-140` and `structure/20_pabcd_dispatch_doctrine.md` -The `components/subagent-config` section does not list `dispatch-contract.ts`; add one line: `- src/dispatch-contract.ts — typed DispatchPacket/DispatchReceipt (#17), verifier coverage (#276) and verifier effects preflight (#277)`. SOT-SYNC-01 target for this phase. +The `components/subagent-config` section of INDEX does not list `dispatch-contract.ts`; add one line: `- src/dispatch-contract.ts — typed DispatchPacket/DispatchReceipt (#17), verifier coverage (#276) and verifier effects preflight (#277)`. `structure/20_pabcd_dispatch_doctrine.md` lists the DISPATCH-* rules (`:120-211`); add a DISPATCH-VERIFIER-01 bullet after DISPATCH-ECONOMY-01 that points to `delegation.md` and names the two functions. These are this phase's SOT-SYNC-01 targets. ## Scope boundary @@ -291,4 +306,3 @@ Red-green: the #276 repro test and the mismatched-command test must fail against - Known bypass: a caller that never calls the functions, or a receipt author who puts an unrelated command's output under a required command's name. - Residual risk: command strings are self-reported; the check proves coverage of names, not that the command ran. - Wording: described as a contract check and preflight, never as enforcement. Final enforcement layer: none. - diff --git a/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md b/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md index 78bd2baa..a336e524 100644 --- a/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md +++ b/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md @@ -44,13 +44,16 @@ one line: require the answer's `eventId`; without one an entry stays `proposed` or `open`, whatever the conversation seemed to imply. Under an active goal, where Interview is suppressed, a decided goalplan decision id is the answer reference. -- Only `proposed` and `open` entries go under `## OPEN ASSUMPTIONS` and into the - tracker's assumptions: freeze carries every recorded tracker assumption into the - manifest as open. Move a `user_confirmed` entry into the plan's requirements with + A reply typed in chat has no `eventId`: confirm it through the next + `request_user_input` round, or keep the entry `open` and quote the reply. +- Only `proposed` and `open` entries go under `## OPEN ASSUMPTIONS`. If a tracker + holds assumptions, the same rule applies there, because freeze carries every + recorded tracker assumption into the manifest as open; do not hand-edit session + state to create one. Move a `user_confirmed` entry into the plan's requirements with its reference. Move a `user_rejected` entry under `## ASSUMPTION DECISIONS` with its reference, so the decision stays traceable without being carried as open. -- A tracker assumption's `text` repeats the same line without its leading `- ` - (freeze prepends it), so the frozen manifest keeps the provenance. +- Where a tracker assumption exists, its `text` repeats the same line without its + leading `- ` (freeze prepends it), so the frozen manifest keeps the provenance. - This rule shapes existing plan text and tracker entries. It adds no field or command, and older plans and trackers read as before. ``` @@ -94,7 +97,7 @@ one line: +existing answer-provenance/readiness and completion gates remain intact. ``` -The file name stays; `minds.test.ts:38` checks that it exists. +The file name stays; `minds.test.ts:37` checks that it exists. ### 3. MODIFY `plugins/codexclaw/skills/loop/references/durable-goalplan.md`, line 40 @@ -112,7 +115,7 @@ SoT sync (SOT-SYNC-01): the Interview skill is the canonical owner of these rule | Check | Where it is met | |---|---| -| 1. inference cannot be presented as user-confirmed without an answer reference | 1(b) status bullet: confirmed/rejected require the answer `eventId` | +| 1. inference cannot be presented as user-confirmed without an answer reference (rule-level; see enforcement naming) | 1(b) status bullet: confirmed/rejected require the answer `eventId` | | 2. rejected inference kept as decision trace, not carried as open | 1(b) last-but-one bullet: `## ASSUMPTION DECISIONS`, never in tracker assumptions | | 3. closeout distinguishes confirmed requirements from open inferred assumptions | 1(d), 1(e) | | 4. existing trackers and freeze manifests remain readable | no code change; 1(b) last bullet | @@ -128,3 +131,6 @@ SoT sync (SOT-SYNC-01): the Interview skill is the canonical owner of these rule No conditional code path is added, so C-ACTIVATION-GROUNDING-01 does not apply; C-READER-01 applies to the new section (a fresh reader checks that the example line and the four statuses are understandable without this doc). +## Enforcement naming (PLAN-BYPASS-NAMED-01) + +Tier E7 (agent-followed guidance). Executing surface: the main session writing the plan. Known bypass: an agent can still label an entry `user_confirmed` without a real `eventId`; nothing checks the reference against the ledger. Residual risk: acceptance check 1 holds by discipline plus reviewability (the reference is visible and checkable in the hashed plan), not by a gate. Wording: guidance, never "cannot"; the acceptance table above reads as "the rule requires". Final enforcement layer: none. diff --git a/devlog/_plan/260930_issue_train/030_wp5_decision_options.md b/devlog/_plan/260930_issue_train/030_wp5_decision_options.md index 2072d80a..aab4a868 100644 --- a/devlog/_plan/260930_issue_train/030_wp5_decision_options.md +++ b/devlog/_plan/260930_issue_train/030_wp5_decision_options.md @@ -6,6 +6,7 @@ A recorded goalplan decision can now carry the options that were offered, and wh - Work phase: `wp5` (goalplan id; runs third, before delivery `wp4`), issue [#262](https://github.com/lidge-jun/codexclaw/issues/262). Class C2 with C4 care for the reviver (a malformed optional field must fail closed, as every existing field does). Design decisions D22-D28 in `002_architect_consultation.md`. - No schema-version bump: `options` is optional and absent on old plans, which round-trip unchanged (`durable-goalplan.md:60`). +- `durable-goalplan.md` is also edited by 020 (three lines added at `:40`); this phase's P re-anchors the `:60` and `:98` edits on the `dev` that contains 020. - Goalplan criterion c-10 records "answer validated against options when present". Architect D25 showed that rule would reject the host's free-form "Other" reply and strand linked phases, so this plan does not implement it; 002 records the disposition and c-10's evidence will state it. ## Field chain (PLAN-FIELD-CHAIN-01) for `GoalplanDecision.options` @@ -14,7 +15,7 @@ A recorded goalplan decision can now carry the options that were offered, and wh |---|---| | Creation | CLI `--option` (`goalplan-cli.ts` parser) -> `runDecision` -> `askGoalplanDecision` input (`goalplan.ts:1225-1256`) | | Serialization | `writeGoalplan` JSON.stringify of the plan object; no custom serializer (field order follows the object literal in `askGoalplanDecision`) | -| Deserialization | `reviveDecisions` (`goalplan.ts:526-550`), also reached by `invalidReason` (`:893`) | +| Deserialization | `reviveDecisions` (`goalplan.ts:526-550`), also reached by `firstInvalidField` (`:857`, the `decisions` check at `:893`) | | Consumers | `ready --json` open-decision projection (`goalplan-cli.ts:465-466`), `show` render (`:688-692`), `decideGoalplanDecision` (`goalplan.ts:1259-1275`: copies the decision with spread, so `options` survives; no membership check per D25), steering/other writers spread the plan and keep `decisions` untouched. Text `ready` (`:499-504`) prints id and question only; unchanged | ## File change map @@ -136,7 +137,8 @@ Beside the existing decision tests (`:684-764`), using the same temp-dir CLI hel | `ask rejects blank and repeated options at parse time` | `--option " "`; `--option A --option A` | exit 1 with the parser reasons, no write | | `ask without --option stores no options key` | plain ask | decision JSON has no `options` property | | `decide keeps options and accepts a free-form answer` | ask with options, `decide --answer "something else"` | exit 0, decision decided, `options` unchanged | -| `reviver fails closed on malformed options` | hand-written plans with `options: []`, `["A","A "]`, `[1]`, and a recommendation outside valid options | the plan reads as invalid naming the `decisions` field (reviver path via `invalidReason`, `goalplan.ts:893`) before `validateGoalplan` runs; the test asserts that message, exact text confirmed at B | +| `reviver fails closed on malformed options` | hand-written plans with `options: {}`, `options: []`, `[" "]`, `["A","A "]`, `[1]`, and a recommendation outside valid options | each read fails naming the field: `field 'decisions' did not satisfy the schema` (`goalplan.ts:719` via `firstInvalidField`), before `validateGoalplan` runs | +| `askGoalplanDecision rejects empty, blank and repeated options (library)` | direct calls with `options: []`, `[" "]`, `["A"," A"]` (import `askGoalplanDecision` from `../src/goalplan.ts`) | `kind:"rejected"` with `must not be empty`, `non-empty text`, `duplicate decision option 'A'`; these branches are library-caller defenses the CLI parser never reaches, so they are driven directly | ### 5. MODIFY docs @@ -153,9 +155,8 @@ Beside the existing decision tests (`:684-764`), using the same temp-dir CLI hel | `npm run build`, `dist-freshness.test.mjs`, `npm test`, `inventory.mjs --check`, `gate.mjs`, `platform-smoke.mjs` | 0 | as in 010 | | docs prose | — | not observed by a command; human review | -Activation scenarios (C-ACTIVATION-GROUNDING-01): each rejection branch in (1c), (1d) and (2d) is driven by a named test above; the free-form `decide` path proves D25. +Activation scenarios (C-ACTIVATION-GROUNDING-01): (2d)'s blank and repeated `--option` rejections and (1d)'s recommendation-membership rejection are driven through the CLI; (1d)'s empty, blank and duplicate checks are reached only by library callers and are driven by the direct `askGoalplanDecision` test; every (1c) reviver branch is driven by hand-written plan files; the free-form `decide` path proves D25. ## Enforcement naming (PLAN-BYPASS-NAMED-01) -Tier E2 (CLI and reviver validation). Executing surface: `cxc loop ask` and every goalplan read. Known bypass: none for stored plans (hand edits fail closed on read); the recommendation check does not prove the question was actually sent with those options. Residual risk: an invalid hand edit makes the whole plan unreadable until repaired, as for every existing field. Wording: validation, not enforcement of what the host displayed. - +Tier E2 (CLI and reviver validation). Executing surface: `cxc loop ask` and every goalplan read. Known bypass: a library caller writing through `writeGoalplan` directly skips `ask`'s checks, but the next read still fails closed on invalid options; nothing proves the question was actually sent with those options. Residual risk: an invalid hand edit makes the whole plan unreadable until repaired, as for every existing field. Wording: validation, not enforcement of what the host displayed. Final enforcement layer: none. diff --git a/devlog/_plan/260930_issue_train/040_wp4_delivery.md b/devlog/_plan/260930_issue_train/040_wp4_delivery.md index 9181ff34..bc7696c1 100644 --- a/devlog/_plan/260930_issue_train/040_wp4_delivery.md +++ b/devlog/_plan/260930_issue_train/040_wp4_delivery.md @@ -5,18 +5,18 @@ This phase lands nothing new in product code. It gets each implementation phase ## Per implementation phase (wp2, wp3, wp5) 1. Branch: wp2 is `codex/issue-train-0930-wp2`, cut from this session's `codex/issue-train-0930` (= `origin/dev` `659de59b` plus this unit's docs). wp3 and wp5 branch from the latest `origin/dev` after the previous PR merges; if the previous PR is still in CI, branch from its phase branch and retarget to `dev` after it merges (the 0927 train's amendment). -2. Local gates at the phase C: `npm run build`; focused tests through `cxc receipt test`; `npm test` (record the TAP total); `node plugins/codexclaw/scripts/inventory.mjs --write --tests ` then `--check --tests `; `node plugins/codexclaw/scripts/gate.mjs`; `node plugins/codexclaw/scripts/platform-smoke.mjs`; `git diff --stat origin/dev -- plugins/codexclaw/hooks plugins/codexclaw/.codex-plugin` shows no hook registration change (criterion c-5). -3. Privacy self-check before the first push (DEV-PRIVACY-01): grep the push range for tokens, client names and home paths other than this user's own (`git log -p origin/dev..HEAD | rg -i 'ghp_|sk-|token=|/Users/(?!jun)'`). -4. `git push -u origin `; `gh pr create --base dev --body-file ` (problem, behavior before/after, tests, residual risk, `Fixes #276` / `Fixes #277` / `Fixes #275` for fully fixed issues; #262 is referenced without `Fixes`). -5. Hosted CI (DEV-CI-EVIDENCE-01): `gh pr view --json headRefOid,statusCheckRollup`, `gh run list --commit --json databaseId,event,headSha,status,conclusion,workflowName`; confirm the ci.yml jobs (ubuntu, macOS, Windows shards, packed install, artifacts), labeler and target check ran on that head, distinguishing pending, skipped, cancelled and failed. Diagnose a failure from its job log before any rerun. +2. Local gates at the phase C, after `npm ci` (without installed dependencies `gui/test/router.test.ts` fails on `react` and the TAP total drops by one; A round 1 measured 3707 against the published 3708): `npm run build`; focused tests through `cxc receipt test`; `npm test` (record the TAP total only from a run with 0 failures); `node plugins/codexclaw/scripts/inventory.mjs --write --tests ` then `--check --tests `; `node plugins/codexclaw/scripts/gate.mjs`; `node plugins/codexclaw/scripts/platform-smoke.mjs`. Criterion c-5: `git diff --stat origin/dev -- plugins/codexclaw/hooks plugins/codexclaw/.codex-plugin plugins/codexclaw/components/pabcd-state/src/hook.ts` is empty and `node plugins/codexclaw/scripts/inventory.mjs --published` still reports 31 hooks. +3. Privacy self-check before the first push (DEV-PRIVACY-01): `git log -p origin/dev..HEAD -- . ':(exclude)devlog/_plan/260930_issue_train/040_wp4_delivery.md' | rg -P -i 'ghp_[A-Za-z0-9]{20,}|sk-[A-Za-z0-9]{20,}|token=[A-Za-z0-9]|/Users/(?!jun/)'` must exit 1 (no match). The exclusion keeps this command's own text from matching. On 2026-09-30 at `048a521c` the unexcluded form matched only this line. +4. `git push -u origin `; `gh pr create --base dev --body-file ` (problem, behavior before/after, tests, residual risk, issue numbers). `Fixes #n` does not close issues from a PR into `dev` because the default branch is `main` (the 0927 train closed #250-#254 by hand); issues are closed manually after merge (below). +5. Hosted CI (DEV-CI-EVIDENCE-01): `gh pr view --json headRefOid,statusCheckRollup`, `gh run list --commit --json databaseId,event,headSha,status,conclusion,workflowName`; confirm the ci.yml jobs (ubuntu, macOS, Windows shards, artifacts), the `packed-install.yml` lifecycle, labeler and target check ran on that head, distinguishing pending, skipped, cancelled and failed. Diagnose a failure from its job log before any rerun. 6. Merge with a merge commit (same method as #269-#272) using `gh pr merge --merge --match-head-commit `. ## Issue disposition (after the implementation PRs merge) -- #275, #276, #277 close through `Fixes`; verify each shows closed with the PR link. +- #275, #276, #277: close as completed with a comment naming the PR and its merge SHA; verify each shows closed. - #262: comment naming the wp5 PR (`options[]` and recommendation membership shipped; answers stay free text by design; `withdrawn` still needs a maintainer decision); stays open. - Close as not planned with the one-line reason from 001 and a link to `devlog/_plan/260930_issue_train/001_research.md` on `dev`: #209, #247, #258, #259, #263, #264, #266, #267, #268. -- Close as completed with the reason and evidence anchors: #213 (plugin scope), #265. +- Close as not planned with the reason and evidence anchors: #213 (the plugin's ownership gate shipped; the remaining atomic authorization is host scope) and #265 (checkpoint guidance shipped; the validator verb is declined). - Comment and keep open: #255, #256, #257, #260, #273, #274. ## Release 0.2.40 @@ -38,4 +38,3 @@ The installed plugin cache and remote hosts are not updated by this train (goal ## Acceptance All goalplan criteria met with captured evidence; `cxc loop validate` passes; v0.2.40 is the latest release and its assets verify. - From 2c322e9ec0be4c4718e1c5268b7fdb2eabb6550b Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Wed, 30 Sep 2026 02:15:49 +0900 Subject: [PATCH 74/90] docs(plan): fold A round 2 residuals --- devlog/_plan/260930_issue_train/000_plan.md | 4 ++-- .../_plan/260930_issue_train/002_architect_consultation.md | 2 +- .../_plan/260930_issue_train/010_wp2_dispatch_contract.md | 7 ++++--- .../260930_issue_train/020_wp3_interview_assumptions.md | 2 +- .../_plan/260930_issue_train/030_wp5_decision_options.md | 4 ++-- 5 files changed, 10 insertions(+), 9 deletions(-) diff --git a/devlog/_plan/260930_issue_train/000_plan.md b/devlog/_plan/260930_issue_train/000_plan.md index 4c1b67a8..799c5121 100644 --- a/devlog/_plan/260930_issue_train/000_plan.md +++ b/devlog/_plan/260930_issue_train/000_plan.md @@ -29,7 +29,7 @@ IN (details in each decade doc): ``` plugins/codexclaw/components/subagent-config/{src,dist,test}/dispatch-contract.* 010 (#276, #277) plugins/codexclaw/skills/pabcd/references/delegation.md 010 -structure/INDEX.md (subagent-config section) 010 SoT sync +structure/INDEX.md (subagent-config section), structure/20_pabcd_dispatch_doctrine.md 010 SoT sync plugins/codexclaw/skills/interview/{SKILL.md,references/mind-dispatch.md} 020 (#275) plugins/codexclaw/skills/loop/references/durable-goalplan.md 020, 030 plugins/codexclaw/components/pabcd-state/{src,dist}/goalplan{,-cli}.*, test/goalplan-public-surface.test.ts 030 (#262) @@ -68,4 +68,4 @@ Delivery: one ordinary PR per implementation phase into `dev`, merged with a mer ## SoT sync targets (SOT-SYNC-01) -`structure/INDEX.md` (subagent-config file list, 010), `skills/interview/SKILL.md` (canonical owner of Interview rules, 020), `skills/loop/references/durable-goalplan.md` (goalplan schema and CLI, 020 and 030), `CHANGELOG.md` (040). +`structure/INDEX.md` (subagent-config file list, 010), `structure/20_pabcd_dispatch_doctrine.md` (DISPATCH-* rule list, 010), `skills/interview/SKILL.md` (canonical owner of Interview rules, 020), `skills/loop/references/durable-goalplan.md` (goalplan schema and CLI, 020 and 030), `CHANGELOG.md` (040). diff --git a/devlog/_plan/260930_issue_train/002_architect_consultation.md b/devlog/_plan/260930_issue_train/002_architect_consultation.md index 5eff9619..c7fc9c5b 100644 --- a/devlog/_plan/260930_issue_train/002_architect_consultation.md +++ b/devlog/_plan/260930_issue_train/002_architect_consultation.md @@ -54,4 +54,4 @@ Main disposition: both folded. 020 1(b) gained the tracker-text bullet; 030's re ## Criterion c-10 correction (A round 1) -Criterion c-10 was registered at this P, before D25 arrived, with the words "answer validated against options when present". D25 showed that rule would reject the host's free-form reply and strand linked phases; neither the user's request nor issue #262 asks for answer membership (the issue asks only that the recommendation be one of the options). Both A auditors flagged that meeting c-10 with an explanatory note would certify text that is not true. Main therefore corrected c-10's scenario to the delivered behavior through a recorded steering `annotate` entry with this rationale, before any implementation. This is a correction of main's own drafting error at P, not a change to a user requirement; it is disclosed in the D summary and the final report. +Criterion c-10 was registered at this P, before D25 arrived, with the words "answer validated against options when present". D25 showed that rule would reject the host's free-form reply and strand linked phases; neither the user's request nor issue #262 asks for answer membership (the issue asks only that the recommendation be one of the options). Both A auditors flagged that meeting c-10 with an explanatory note would certify text that is not true. Main therefore corrected c-10's scenario text by hand edit, with the rationale recorded as steering `annotate` entry `wp1-a1-folds` (annotate is ledger-only; the old text survives in the original `add-criterion` ledger event), before any implementation. This is a correction of main's own drafting error at P, not a change to a user requirement; it is disclosed in the D summary and the final report. diff --git a/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md b/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md index b843e3d2..a3e7fc08 100644 --- a/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md +++ b/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md @@ -165,7 +165,8 @@ export function receiptSatisfiesPacket(packet: DispatchPacket, receipt: Dispatch ]; // Unvalidated input must not throw here; malformed entries fail the receipt. const results = reportedResults.filter(isVerifierResult); - if (results.length !== reportedResults.length) { + const nonArrayResults = receipt.verifierResults !== undefined && !Array.isArray(receipt.verifierResults); + if (nonArrayResults || results.length !== reportedResults.length) { reasons.push("receipt has malformed verifier results (see validateReceipt)"); } if (required.length > 0 && results.length === 0) { @@ -285,7 +286,7 @@ The `components/subagent-config` section of INDEX does not list `dispatch-contra ## Scope boundary -IN: the five files above. OUT: `commandsRun` semantics, `sourceIdentity`, release-gate's `dispatch-contracts` receipt (`pabcd-state/src/release-gate.ts:393`, stays missing), #256/#273 receipt fields, any hook. +IN: the six files above (including `structure/20_pabcd_dispatch_doctrine.md`). OUT: `commandsRun` semantics, `sourceIdentity`, release-gate's `dispatch-contracts` receipt (`pabcd-state/src/release-gate.ts:393`, stays missing), #256/#273 receipt fields, any hook. ## Verification (PLAN-VERIFIER-REAL-01) @@ -294,7 +295,7 @@ IN: the five files above. OUT: `commandsRun` semantics, `sourceIdentity`, releas | `node plugins/codexclaw/scripts/test.mjs plugins/codexclaw/components/subagent-config/test/dispatch-contract.test.ts` | 0 (17 pass) | yes: the file is the direct argument and imports `../src/dispatch-contract.ts` | | `npm run build` | 0 | yes: `build.mjs` recompiles every component `src` into `dist` | | `node --test plugins/codexclaw/test/dist-freshness.test.mjs` | 0 | yes: compares tracked `dist/dispatch-contract.js` with the compiled source | -| `npm test` then `node plugins/codexclaw/scripts/inventory.mjs --check --tests ` | 0 | yes: root glob includes `subagent-config/test/*.test.ts`; inventory checks the README badge total | +| `npm test` (after `npm ci`) then `node plugins/codexclaw/scripts/inventory.mjs --check --tests ` | 0 | yes: root glob includes `subagent-config/test/*.test.ts`; inventory checks the README badge total | | `node plugins/codexclaw/scripts/gate.mjs`, `node plugins/codexclaw/scripts/platform-smoke.mjs` | 0 | gate: inventory and skill checks; smoke: packaging. Neither observes the new logic; they guard regressions only | | delegation.md prose | — | this command does not observe this change; human review in A and C | diff --git a/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md b/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md index a336e524..44c57af5 100644 --- a/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md +++ b/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md @@ -126,7 +126,7 @@ SoT sync (SOT-SYNC-01): the Interview skill is the canonical owner of these rule |---|---|---| | `rg -n 'INTERVIEW-ASSUME-01' plugins/codexclaw/skills` | 1 before (no match), expected 0 after with hits in the three files | yes: the three paths are under the searched directory | | `node plugins/codexclaw/scripts/gate.mjs` | 0 | partly: skill frontmatter/inventory checks read `skills/interview/SKILL.md`; it does not read rule prose | -| `npm test` | 0 | no test reads these prose lines (explorer search of the rule strings found hits only in the SKILL.md and `hook.ts:1923,1932`); run as a regression guard | +| `npm test` (after `npm ci`) | 0 | no test reads these prose lines (explorer search of the rule strings found hits only in the SKILL.md and `hook.ts:1923,1932`); run as a regression guard | | prose meaning | — | this command does not observe this change; human review in A (reviewer) and C (initiative verifier) | No conditional code path is added, so C-ACTIVATION-GROUNDING-01 does not apply; C-READER-01 applies to the new section (a fresh reader checks that the example line and the four statuses are understandable without this doc). diff --git a/devlog/_plan/260930_issue_train/030_wp5_decision_options.md b/devlog/_plan/260930_issue_train/030_wp5_decision_options.md index aab4a868..952ecf68 100644 --- a/devlog/_plan/260930_issue_train/030_wp5_decision_options.md +++ b/devlog/_plan/260930_issue_train/030_wp5_decision_options.md @@ -7,7 +7,7 @@ A recorded goalplan decision can now carry the options that were offered, and wh - Work phase: `wp5` (goalplan id; runs third, before delivery `wp4`), issue [#262](https://github.com/lidge-jun/codexclaw/issues/262). Class C2 with C4 care for the reviver (a malformed optional field must fail closed, as every existing field does). Design decisions D22-D28 in `002_architect_consultation.md`. - No schema-version bump: `options` is optional and absent on old plans, which round-trip unchanged (`durable-goalplan.md:60`). - `durable-goalplan.md` is also edited by 020 (three lines added at `:40`); this phase's P re-anchors the `:60` and `:98` edits on the `dev` that contains 020. -- Goalplan criterion c-10 records "answer validated against options when present". Architect D25 showed that rule would reject the host's free-form "Other" reply and strand linked phases, so this plan does not implement it; 002 records the disposition and c-10's evidence will state it. +- Answers stay free text (architect D25): an answer-in-options rule would reject the host's free-form "Other" reply and strand linked phases. Criterion c-10 was corrected at P to match; see 002, "Criterion c-10 correction". ## Field chain (PLAN-FIELD-CHAIN-01) for `GoalplanDecision.options` @@ -152,7 +152,7 @@ Beside the existing decision tests (`:684-764`), using the same temp-dir CLI hel |---|---|---| | `node plugins/codexclaw/scripts/test.mjs plugins/codexclaw/components/pabcd-state/test/goalplan-public-surface.test.ts` | 0 | yes: drives the CLI and imports `../src/goalplan*.ts` | | `node plugins/codexclaw/scripts/test.mjs plugins/codexclaw/components/pabcd-state/test/goalplan.test.ts plugins/codexclaw/components/pabcd-state/test/goalplan-integrity.test.ts plugins/codexclaw/components/pabcd-state/test/hook-continuation.test.ts` | 0 | yes: round-trip, integrity and Stop decision-wait paths read the reviver | -| `npm run build`, `dist-freshness.test.mjs`, `npm test`, `inventory.mjs --check`, `gate.mjs`, `platform-smoke.mjs` | 0 | as in 010 | +| `npm run build`, `dist-freshness.test.mjs`, `npm test` (after `npm ci`), `inventory.mjs --check`, `gate.mjs`, `platform-smoke.mjs` | 0 | as in 010 | | docs prose | — | not observed by a command; human review | Activation scenarios (C-ACTIVATION-GROUNDING-01): (2d)'s blank and repeated `--option` rejections and (1d)'s recommendation-membership rejection are driven through the CLI; (1d)'s empty, blank and duplicate checks are reached only by library callers and are driven by the direct `askGoalplanDecision` test; every (1c) reviver branch is driven by hand-written plan files; the free-form `decide` path proves D25. From a6abcd1258d3fe2881f017d2c960f44060ec0b30 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Wed, 30 Sep 2026 02:16:28 +0900 Subject: [PATCH 75/90] docs(plan): record wp1 audit rounds --- devlog/_plan/260930_issue_train/000_plan.md | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/devlog/_plan/260930_issue_train/000_plan.md b/devlog/_plan/260930_issue_train/000_plan.md index 799c5121..9cd837ff 100644 --- a/devlog/_plan/260930_issue_train/000_plan.md +++ b/devlog/_plan/260930_issue_train/000_plan.md @@ -69,3 +69,13 @@ Delivery: one ordinary PR per implementation phase into `dev`, merged with a mer ## SoT sync targets (SOT-SYNC-01) `structure/INDEX.md` (subagent-config file list, 010), `structure/20_pabcd_dispatch_doctrine.md` (DISPATCH-* rule list, 010), `skills/interview/SKILL.md` (canonical owner of Interview rules, 020), `skills/loop/references/durable-goalplan.md` (goalplan schema and CLI, 020 and 030), `CHANGELOG.md` (040). + +## Review record + +wp1 (this roadmap), 2026-09-30. Reviewer `01a0ee1f-b88c-73d0-ae1a-c526f311d2c2` (cxc-dev-code-reviewer, cxc-search) and PABCD-initiative verifier `01a0ee1f-b983-7e61-b27e-c9eeeeada686` (canonical `pabcd_initiative/skills/dev-pabcd/SKILL.md`) ran in parallel, both inheriting this session's model. + +- Round 1: both GO-WITH-FIXES with five blockers each, largely overlapping: criterion c-10 contradicted architect D25; the privacy grep errored (`rg` without PCRE2); rejection branches claimed as tested were unreachable from the CLI; 010 lacked a field-chain table and 030 a final enforcement layer; local `npm test` totals were wrong without `npm ci`; `Fixes #n` does not close issues from PRs into `dev`. The initiative verifier's rule table marked LEXICO-SPLIT-01, UNIT-RESIDENCE-01, LOOP-DOCS-FIRST-01, the loop-spec fields and the reader summary compliant, and REVIEW-DECORRELATE-01 not established (disclosed in the loop contract). +- Folds in `dfdb4b01`; rebuttals: none. +- Round 2 (same agents): both PASS. Residuals (stale c-10 wording in 030, 002's description of the hand edit, the six-file scope, `structure/20` in the SoT list, the `npm ci` caveat, a malformed-reason for a non-array `verifierResults`) folded in `2c322e9e`. +- Port vs canonical notes from the initiative verifier: the canonical rule asks for token and wall-clock bounds on C4 work while the codexclaw port forbids inventing them (disclosed above); canonical entry-edge attests versus the port's attest-free entry edges (no practical effect); canonical full-plan injection to workers versus the port's path pointers (read-only reviews unaffected). + From 23eba83e8dade3ba001fb9390ef269c0efcfd45e Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Wed, 30 Sep 2026 02:17:32 +0900 Subject: [PATCH 76/90] docs(plan): wp2 P revalidation --- devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md b/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md index a3e7fc08..a8317814 100644 --- a/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md +++ b/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md @@ -307,3 +307,8 @@ Red-green: the #276 repro test and the mismatched-command test must fail against - Known bypass: a caller that never calls the functions, or a receipt author who puts an unrelated command's output under a required command's name. - Residual risk: command strings are self-reported; the check proves coverage of names, not that the command ran. - Wording: described as a contract check and preflight, never as enforcement. Final enforcement layer: none. + +## wp2 P revalidation (2026-09-30) + +Continuity: wp1 D locked this doc as the wp2 plan (roadmap audited PASS twice). Re-checked on `codex/issue-train-0930-wp2`: `git diff --stat 659de59b..HEAD -- plugins structure` is empty, so every anchor above still holds. One amendment: the `structure/INDEX.md` subagent-config section (`:138-140`) is a prose paragraph, so §5 appends one sentence to it: "`src/dispatch-contract.ts` holds the typed DispatchPacket/DispatchReceipt contract (#17): verifier coverage for receipts (#276) and a pure verifier-effects preflight (#277); no runtime path calls it yet." The `structure/20` bullet goes after the DISPATCH-ECONOMY-01 bullet's last clause (`:253`, "from mechanics to economy."), before the `---` that closes §3. Architect consultation: this is the same plan the architect proposed and reflected on (002); the amendment changes no design decision, so no re-consultation (phase-audit "Architect recheck" trigger not met). + From 8a4b34046491f9d4c9fe5dd879f14b1d6c8aa297 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Wed, 30 Sep 2026 02:19:11 +0900 Subject: [PATCH 77/90] docs(plan): wp1 D summary; fold wp2 A round 1 --- devlog/_plan/260930_issue_train/000_plan.md | 6 ++++++ .../260930_issue_train/010_wp2_dispatch_contract.md | 9 ++++++--- 2 files changed, 12 insertions(+), 3 deletions(-) diff --git a/devlog/_plan/260930_issue_train/000_plan.md b/devlog/_plan/260930_issue_train/000_plan.md index 9cd837ff..f07e77a8 100644 --- a/devlog/_plan/260930_issue_train/000_plan.md +++ b/devlog/_plan/260930_issue_train/000_plan.md @@ -79,3 +79,9 @@ wp1 (this roadmap), 2026-09-30. Reviewer `01a0ee1f-b88c-73d0-ae1a-c526f311d2c2` - Round 2 (same agents): both PASS. Residuals (stale c-10 wording in 030, 002's description of the hand edit, the six-file scope, `structure/20` in the SoT list, the `npm ci` caveat, a malformed-reason for a non-array `verifierResults`) folded in `2c322e9e`. - Port vs canonical notes from the initiative verifier: the canonical rule asks for token and wall-clock bounds on C4 work while the codexclaw port forbids inventing them (disclosed above); canonical entry-edge attests versus the port's attest-free entry edges (no practical effect); canonical full-plan injection to workers versus the port's path pointers (read-only reviews unaffected). + +## wp1 D summary (2026-09-30) + +Conclusion: the roadmap is locked and wp2 builds 010 as written; wp3 builds 020, wp5 builds 030, and wp4 delivers per 040. Evidence: two audit rounds ending in PASS from both agents (Review record above), the wp1 C receipt (lexico naming, docs-only scope, `gate.mjs` OK) and a baseline `npm test` after `npm ci`: 3708 tests, 3703 pass, 0 fail, 5 skipped, exit 0, matching the published badge. + +What did not go well: the first plan draft claimed tests for branches the CLI cannot reach, shipped a privacy grep that did not run, and registered criterion c-10 with a rule the architect then rejected, which needed a disclosed correction. Review independence is context-only because every agent inherits this session's model, and wp4 (a C4 release) runs without a stated token or wall-clock bound. Evidence that this direction is wrong would be a reviewer finding that the strict unrelated-result rule of #276 breaks a real caller (none exists today), or hosted CI failing on a platform difference the local run hides. diff --git a/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md b/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md index a8317814..d56b086c 100644 --- a/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md +++ b/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md @@ -258,7 +258,7 @@ Import `verifierPreflight` beside the existing imports (`:6-12`). Existing tests | `validatePacket: rejects blank verifier command entries` | `verifierCommands:["npm test"," "]` | error `entries must be non-empty strings` | | `validatePacket: verifierEffects shape` | unknown command; duplicate command; `expectedWrites:"x"`; `runInIsolation:"yes"`; valid `[{command:"npm test",expectedWrites:[]}]` | four errors, then `[]` | | `validatePacket: verifierEffects container and entry guards` | `verifierEffects:{}`; `verifierEffects:[null]`; `verifierEffects:[{command:" ",expectedWrites:[]}]` | errors `must be an array`, `entries must be objects`, `command must be a non-empty string` | -| `receiptSatisfiesPacket: malformed unvalidated result does not throw` | `verifierResults:[{command:1}]` cast past the type | returns `satisfied:false` with the `malformed verifier results` reason | +| `receiptSatisfiesPacket: malformed unvalidated result does not throw` | `verifierResults:[{command:1}]` and `verifierResults:{}`, each cast past the type | both return `satisfied:false` with the `malformed verifier results` reason (the second drives the non-array branch) | | `verifierPreflight: shared-read flags undeclared and writing verifiers` | shared-read with undeclared, `[]`, writes, `runInIsolation` | `needsIsolation` true/false/true/true, `declared` false/true/true/true | | `verifierPreflight: isolated-write accepts declared writes` | isolated-write with writes | `needsIsolation:false`, `declared:true` | @@ -282,7 +282,10 @@ A declaration is the author's claim, not proof: codexclaw never executes it. ### 5. SoT sync `structure/INDEX.md:138-140` and `structure/20_pabcd_dispatch_doctrine.md` -The `components/subagent-config` section of INDEX does not list `dispatch-contract.ts`; add one line: `- src/dispatch-contract.ts — typed DispatchPacket/DispatchReceipt (#17), verifier coverage (#276) and verifier effects preflight (#277)`. `structure/20_pabcd_dispatch_doctrine.md` lists the DISPATCH-* rules (`:120-211`); add a DISPATCH-VERIFIER-01 bullet after DISPATCH-ECONOMY-01 that points to `delegation.md` and names the two functions. These are this phase's SOT-SYNC-01 targets. +- `structure/INDEX.md`: the `components/subagent-config` section (`:138-140`) is one prose paragraph; append the sentence "`src/dispatch-contract.ts` holds the typed DispatchPacket/DispatchReceipt contract (#17): verifier coverage for receipts (#276) and a pure verifier-effects preflight (#277); no runtime path calls it yet." +- `structure/20_pabcd_dispatch_doctrine.md`: after the DISPATCH-ECONOMY-01 bullet's last clause (`:253`, "from mechanics to economy."), before the `---` closing section 3 (`:255`), add a **DISPATCH-VERIFIER-01** bullet: E2 library contract (`receiptSatisfiesPacket`, `verifierPreflight` in `components/subagent-config/src/dispatch-contract.ts`) plus E7 guidance in `skills/pabcd/references/delegation.md`; no hook calls it. + +These are this phase's SOT-SYNC-01 targets. ## Scope boundary @@ -310,5 +313,5 @@ Red-green: the #276 repro test and the mismatched-command test must fail against ## wp2 P revalidation (2026-09-30) -Continuity: wp1 D locked this doc as the wp2 plan (roadmap audited PASS twice). Re-checked on `codex/issue-train-0930-wp2`: `git diff --stat 659de59b..HEAD -- plugins structure` is empty, so every anchor above still holds. One amendment: the `structure/INDEX.md` subagent-config section (`:138-140`) is a prose paragraph, so §5 appends one sentence to it: "`src/dispatch-contract.ts` holds the typed DispatchPacket/DispatchReceipt contract (#17): verifier coverage for receipts (#276) and a pure verifier-effects preflight (#277); no runtime path calls it yet." The `structure/20` bullet goes after the DISPATCH-ECONOMY-01 bullet's last clause (`:253`, "from mechanics to economy."), before the `---` that closes §3. Architect consultation: this is the same plan the architect proposed and reflected on (002); the amendment changes no design decision, so no re-consultation (phase-audit "Architect recheck" trigger not met). +Continuity (LOOP-CONTINUITY-01), quoting the wp1 D summary in 000: "the roadmap is locked and wp2 builds 010 as written"; its negative side (context-only review independence, no stated resource bound for wp4, the c-10 correction) is carried forward. This P keeps that direction. Re-checked on `codex/issue-train-0930-wp2`: `git diff --stat 659de59b..HEAD -- plugins structure` is empty, so every anchor above still holds. One amendment: the `structure/INDEX.md` subagent-config section (`:138-140`) is a prose paragraph, so §5 appends one sentence to it: "`src/dispatch-contract.ts` holds the typed DispatchPacket/DispatchReceipt contract (#17): verifier coverage for receipts (#276) and a pure verifier-effects preflight (#277); no runtime path calls it yet." The `structure/20` bullet goes after the DISPATCH-ECONOMY-01 bullet's last clause (`:253`, "from mechanics to economy."), before the `---` that closes §3. Architect consultation: after the reflection at `048a521c`, A folds added a fail-closed malformed-result branch to (1g), including the non-array case; it implements D4 (every result must be a passing, matching result) and D8 (both result fields are shape-checked), so it changes no design decision. This amendment changes placement only, so the phase-audit architect-recheck trigger is not met. From b852eb24fe86017bd203dfff19f21b88ce18fd29 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Wed, 30 Sep 2026 02:21:04 +0900 Subject: [PATCH 78/90] fix(dispatch): require a matching verifier result per required command; add optional verifier effects preflight (#276, #277) --- .../subagent-config/dist/dispatch-contract.js | 159 +++++++++++++++- .../subagent-config/src/dispatch-contract.ts | 163 ++++++++++++++++- .../test/dispatch-contract.test.ts | 171 ++++++++++++++++++ .../skills/pabcd/references/delegation.md | 9 + structure/20_pabcd_dispatch_doctrine.md | 7 + structure/INDEX.md | 2 +- 6 files changed, 498 insertions(+), 13 deletions(-) diff --git a/plugins/codexclaw/components/subagent-config/dist/dispatch-contract.js b/plugins/codexclaw/components/subagent-config/dist/dispatch-contract.js index 2b2532e5..a779251a 100644 --- a/plugins/codexclaw/components/subagent-config/dist/dispatch-contract.js +++ b/plugins/codexclaw/components/subagent-config/dist/dispatch-contract.js @@ -13,6 +13,27 @@ import { ROLES, } from "./store.js"; /** Terminal status of a dispatched subagent. */ +/** One verifier command's result as reported by the subagent (#276). */ + + + + + + +/** + * Declared write effects of one verifier command (#277). A declaration is the + * packet author's claim, not proof: codexclaw never runs the command and does + * not check paths against a filesystem. + */ + + + + + + + + + /** * DispatchPacket — structured task specification for a subagent. * Contains everything a subagent needs to complete its bounded task. @@ -37,6 +58,8 @@ import { ROLES, } from "./store.js"; + + @@ -63,6 +86,51 @@ import { ROLES, } from "./store.js"; + + +function isVerifierResult(value ) { + if (!value || typeof value !== "object" || Array.isArray(value)) return false; + const v = value ; + return typeof v.command === "string" && Number.isInteger(v.exitCode) && typeof v.output === "string"; +} + +/** Distinct trimmed commands; exact string equality after trim, no other normalization. */ +function distinctCommands(commands ) { + return [...new Set(commands.map((command) => command.trim()))]; +} + +function validateVerifierEffects(value , commands ) { + if (!Array.isArray(value)) return ["verifierEffects must be an array"]; + const required = new Set(Array.isArray(commands) + ? commands.filter((command) => typeof command === "string").map((command) => command.trim()) + : []); + const seen = new Set (); + const errors = []; + for (const item of value) { + if (!item || typeof item !== "object" || Array.isArray(item)) { + errors.push("verifierEffects entries must be objects"); + continue; + } + const effect = item ; + if (typeof effect.command !== "string" || !effect.command.trim()) { + errors.push("verifierEffects command must be a non-empty string"); + continue; + } + const command = effect.command.trim(); + if (!required.has(command)) errors.push("verifierEffects command `" + command + "` is not in verifierCommands"); + if (seen.has(command)) errors.push("verifierEffects declares `" + command + "` more than once"); + seen.add(command); + if (!Array.isArray(effect.expectedWrites) + || effect.expectedWrites.some((path) => typeof path !== "string" || !path.trim())) { + errors.push("verifierEffects expectedWrites for `" + command + "` must be an array of non-empty strings"); + } + if (effect.runInIsolation !== undefined && typeof effect.runInIsolation !== "boolean") { + errors.push("verifierEffects runInIsolation for `" + command + "` must be a boolean"); + } + } + return errors; +} + /** Validate a DispatchPacket. Returns error messages or empty array. */ export function validatePacket(packet ) { const errors = []; @@ -76,6 +144,10 @@ export function validatePacket(packet ) { if (typeof p.expectedOutput !== "string") errors.push("expectedOutput must be a string"); if (typeof p.decisionBoundary !== "string") errors.push("decisionBoundary must be a string"); if (!Array.isArray(p.verifierCommands)) errors.push("verifierCommands must be an array"); + else if (p.verifierCommands.some((command) => typeof command !== "string" || !command.trim())) { + errors.push("verifierCommands entries must be non-empty strings"); + } + if (p.verifierEffects !== undefined) errors.push(...validateVerifierEffects(p.verifierEffects, p.verifierCommands)); if (!Array.isArray(p.requiredSkills)) errors.push("requiredSkills must be an array"); if (p.worktreePolicy !== "shared-read" && p.worktreePolicy !== "isolated-write") { errors.push("worktreePolicy must be shared-read or isolated-write"); @@ -106,13 +178,26 @@ export function validateReceipt(receipt ) { if (!Array.isArray(r.evidenceAnchors)) errors.push("evidenceAnchors must be an array"); if (!Array.isArray(r.commandsRun)) errors.push("commandsRun must be an array"); if (!Array.isArray(r.unresolvedAssumptions)) errors.push("unresolvedAssumptions must be an array"); + if (r.verifierResult !== undefined && !isVerifierResult(r.verifierResult)) { + errors.push("verifierResult must be {command: string, exitCode: integer, output: string}"); + } + if (r.verifierResults !== undefined + && (!Array.isArray(r.verifierResults) || !r.verifierResults.every(isVerifierResult))) { + errors.push("verifierResults must be an array of {command: string, exitCode: integer, output: string}"); + } return errors; } -/** Check that a receipt satisfies its packet's verifier requirements. */ +/** + * Check that a receipt satisfies its packet's verifier requirements (#276). + * Every distinct required command needs a matching result, every result must + * exit 0, and a result for a command the packet did not require is rejected. + */ export function receiptSatisfiesPacket(packet , receipt ) + + { const reasons = []; if (receipt.packetId !== packet.id) { @@ -121,12 +206,76 @@ export function receiptSatisfiesPacket(packet , receipt if (receipt.status !== "complete") { reasons.push("receipt status is " + receipt.status + ", not complete"); } - if (packet.verifierCommands.length > 0 && !receipt.verifierResult) { + const required = distinctCommands(packet.verifierCommands); + const reportedResults = [ + ...(Array.isArray(receipt.verifierResults) ? receipt.verifierResults : []), + ...(receipt.verifierResult ? [receipt.verifierResult] : []), + ]; + // Unvalidated input must not throw here; malformed entries fail the receipt. + const results = reportedResults.filter(isVerifierResult); + const nonArrayResults = receipt.verifierResults !== undefined && !Array.isArray(receipt.verifierResults); + if (nonArrayResults || results.length !== reportedResults.length) { + reasons.push("receipt has malformed verifier results (see validateReceipt)"); + } + if (required.length > 0 && results.length === 0) { reasons.push("packet has verifier commands but receipt has no verifier result"); } - if (receipt.verifierResult && receipt.verifierResult.exitCode !== 0) { - reasons.push("verifier exit code " + receipt.verifierResult.exitCode + " (expected 0)"); + for (const result of results) { + if (result.exitCode !== 0) { + reasons.push("verifier exit code " + result.exitCode + " (expected 0) for `" + result.command.trim() + "`"); + } + } + const reported = new Set(results.map((result) => result.command.trim())); + const missing = required.filter((command) => !reported.has(command)); + if (results.length > 0) { + for (const command of missing) reasons.push("missing verifier result for `" + command + "`"); } - return { satisfied: reasons.length === 0, reasons }; + if (required.length > 1 && receipt.verifierResults === undefined && receipt.verifierResult) { + reasons.push("receipt reports one legacy verifierResult; packet requires " + + required.length + " verifier commands (incomplete)"); + } + if (required.length > 0) { + const requiredSet = new Set(required); + for (const command of reported) { + if (!requiredSet.has(command)) reasons.push("verifier result for unrelated command `" + command + "`"); + } + } + return { satisfied: reasons.length === 0, reasons, missing }; +} + +/** One preflight row per distinct verifier command (#277). */ + + + + + + + +/** + * Pure preflight over declared verifier effects (#277). It never runs a command + * or reads the filesystem; it only tells the caller which verifiers must not run + * on a shared checkout without isolation or main's confirmation. + */ +export function verifierPreflight(packet ) { + const effects = new Map((packet.verifierEffects ?? []).map((effect) => [effect.command.trim(), effect])); + return distinctCommands(packet.verifierCommands).map((command) => { + const effect = effects.get(command); + const declared = effect !== undefined; + if (packet.worktreePolicy === "isolated-write") { + return { command, declared, needsIsolation: false, reason: "packet is isolated-write" }; + } + if (!effect) { + return { command, declared, needsIsolation: true, + reason: "no declared write boundary on a shared-read packet; run it in an isolated copy or confirm with main" }; + } + if (effect.runInIsolation === true) { + return { command, declared, needsIsolation: true, reason: "declared runInIsolation" }; + } + if (effect.expectedWrites.length > 0) { + return { command, declared, needsIsolation: true, + reason: "declares writes (" + effect.expectedWrites.join(", ") + ") on a shared-read packet" }; + } + return { command, declared, needsIsolation: false, reason: "declared read-only (expectedWrites: [])" }; + }); } diff --git a/plugins/codexclaw/components/subagent-config/src/dispatch-contract.ts b/plugins/codexclaw/components/subagent-config/src/dispatch-contract.ts index 7ac38a27..6fccc893 100644 --- a/plugins/codexclaw/components/subagent-config/src/dispatch-contract.ts +++ b/plugins/codexclaw/components/subagent-config/src/dispatch-contract.ts @@ -13,6 +13,27 @@ export type WorktreePolicy = "shared-read" | "isolated-write"; /** Terminal status of a dispatched subagent. */ export type DispatchStatus = "complete" | "blocked" | "inconclusive"; +/** One verifier command's result as reported by the subagent (#276). */ +export interface VerifierResult { + command: string; + exitCode: number; + output: string; +} + +/** + * Declared write effects of one verifier command (#277). A declaration is the + * packet author's claim, not proof: codexclaw never runs the command and does + * not check paths against a filesystem. + */ +export interface VerifierEffect { + /** Must equal (after trim) one entry of `verifierCommands`. */ + command: string; + /** Paths or globs the command may write; `[]` declares it read-only. */ + expectedWrites: string[]; + /** Run this verifier in an isolated copy even when the packet is shared-read. */ + runInIsolation?: boolean; +} + /** * DispatchPacket — structured task specification for a subagent. * Contains everything a subagent needs to complete its bounded task. @@ -30,6 +51,8 @@ export interface DispatchPacket { decisionBoundary: string; /** Verifier commands the main agent will run to check the result. */ verifierCommands: string[]; + /** Optional write-effect declarations, at most one per verifier command (#277). */ + verifierEffects?: VerifierEffect[]; /** Skill names to attach to the subagent. */ requiredSkills: string[]; /** Worktree access policy. */ @@ -57,12 +80,57 @@ export interface DispatchReceipt { commandsRun: string[]; /** Unresolved assumptions the main agent must evaluate. */ unresolvedAssumptions: string[]; - /** Verifier result from the subagent's perspective. */ - verifierResult?: { command: string; exitCode: number; output: string }; + /** Legacy single verifier result; still read, merged with `verifierResults`. */ + verifierResult?: VerifierResult; + /** One result per packet verifier command (#276). Extra checks go in `commandsRun`. */ + verifierResults?: VerifierResult[]; /** Source/worktree identity where mutation occurred. */ sourceIdentity?: string; } +function isVerifierResult(value: unknown): value is VerifierResult { + if (!value || typeof value !== "object" || Array.isArray(value)) return false; + const v = value as Record; + return typeof v.command === "string" && Number.isInteger(v.exitCode) && typeof v.output === "string"; +} + +/** Distinct trimmed commands; exact string equality after trim, no other normalization. */ +function distinctCommands(commands: readonly string[]): string[] { + return [...new Set(commands.map((command) => command.trim()))]; +} + +function validateVerifierEffects(value: unknown, commands: unknown): string[] { + if (!Array.isArray(value)) return ["verifierEffects must be an array"]; + const required = new Set(Array.isArray(commands) + ? commands.filter((command): command is string => typeof command === "string").map((command) => command.trim()) + : []); + const seen = new Set(); + const errors: string[] = []; + for (const item of value) { + if (!item || typeof item !== "object" || Array.isArray(item)) { + errors.push("verifierEffects entries must be objects"); + continue; + } + const effect = item as Record; + if (typeof effect.command !== "string" || !effect.command.trim()) { + errors.push("verifierEffects command must be a non-empty string"); + continue; + } + const command = effect.command.trim(); + if (!required.has(command)) errors.push("verifierEffects command `" + command + "` is not in verifierCommands"); + if (seen.has(command)) errors.push("verifierEffects declares `" + command + "` more than once"); + seen.add(command); + if (!Array.isArray(effect.expectedWrites) + || effect.expectedWrites.some((path) => typeof path !== "string" || !path.trim())) { + errors.push("verifierEffects expectedWrites for `" + command + "` must be an array of non-empty strings"); + } + if (effect.runInIsolation !== undefined && typeof effect.runInIsolation !== "boolean") { + errors.push("verifierEffects runInIsolation for `" + command + "` must be a boolean"); + } + } + return errors; +} + /** Validate a DispatchPacket. Returns error messages or empty array. */ export function validatePacket(packet: unknown): string[] { const errors: string[] = []; @@ -76,6 +144,10 @@ export function validatePacket(packet: unknown): string[] { if (typeof p.expectedOutput !== "string") errors.push("expectedOutput must be a string"); if (typeof p.decisionBoundary !== "string") errors.push("decisionBoundary must be a string"); if (!Array.isArray(p.verifierCommands)) errors.push("verifierCommands must be an array"); + else if (p.verifierCommands.some((command) => typeof command !== "string" || !command.trim())) { + errors.push("verifierCommands entries must be non-empty strings"); + } + if (p.verifierEffects !== undefined) errors.push(...validateVerifierEffects(p.verifierEffects, p.verifierCommands)); if (!Array.isArray(p.requiredSkills)) errors.push("requiredSkills must be an array"); if (p.worktreePolicy !== "shared-read" && p.worktreePolicy !== "isolated-write") { errors.push("worktreePolicy must be shared-read or isolated-write"); @@ -106,13 +178,26 @@ export function validateReceipt(receipt: unknown): string[] { if (!Array.isArray(r.evidenceAnchors)) errors.push("evidenceAnchors must be an array"); if (!Array.isArray(r.commandsRun)) errors.push("commandsRun must be an array"); if (!Array.isArray(r.unresolvedAssumptions)) errors.push("unresolvedAssumptions must be an array"); + if (r.verifierResult !== undefined && !isVerifierResult(r.verifierResult)) { + errors.push("verifierResult must be {command: string, exitCode: integer, output: string}"); + } + if (r.verifierResults !== undefined + && (!Array.isArray(r.verifierResults) || !r.verifierResults.every(isVerifierResult))) { + errors.push("verifierResults must be an array of {command: string, exitCode: integer, output: string}"); + } return errors; } -/** Check that a receipt satisfies its packet's verifier requirements. */ +/** + * Check that a receipt satisfies its packet's verifier requirements (#276). + * Every distinct required command needs a matching result, every result must + * exit 0, and a result for a command the packet did not require is rejected. + */ export function receiptSatisfiesPacket(packet: DispatchPacket, receipt: DispatchReceipt): { satisfied: boolean; reasons: string[]; + /** Required verifier commands (trimmed) with no matching result. */ + missing: string[]; } { const reasons: string[] = []; if (receipt.packetId !== packet.id) { @@ -121,12 +206,76 @@ export function receiptSatisfiesPacket(packet: DispatchPacket, receipt: Dispatch if (receipt.status !== "complete") { reasons.push("receipt status is " + receipt.status + ", not complete"); } - if (packet.verifierCommands.length > 0 && !receipt.verifierResult) { + const required = distinctCommands(packet.verifierCommands); + const reportedResults: unknown[] = [ + ...(Array.isArray(receipt.verifierResults) ? receipt.verifierResults : []), + ...(receipt.verifierResult ? [receipt.verifierResult] : []), + ]; + // Unvalidated input must not throw here; malformed entries fail the receipt. + const results = reportedResults.filter(isVerifierResult); + const nonArrayResults = receipt.verifierResults !== undefined && !Array.isArray(receipt.verifierResults); + if (nonArrayResults || results.length !== reportedResults.length) { + reasons.push("receipt has malformed verifier results (see validateReceipt)"); + } + if (required.length > 0 && results.length === 0) { reasons.push("packet has verifier commands but receipt has no verifier result"); } - if (receipt.verifierResult && receipt.verifierResult.exitCode !== 0) { - reasons.push("verifier exit code " + receipt.verifierResult.exitCode + " (expected 0)"); + for (const result of results) { + if (result.exitCode !== 0) { + reasons.push("verifier exit code " + result.exitCode + " (expected 0) for `" + result.command.trim() + "`"); + } + } + const reported = new Set(results.map((result) => result.command.trim())); + const missing = required.filter((command) => !reported.has(command)); + if (results.length > 0) { + for (const command of missing) reasons.push("missing verifier result for `" + command + "`"); + } + if (required.length > 1 && receipt.verifierResults === undefined && receipt.verifierResult) { + reasons.push("receipt reports one legacy verifierResult; packet requires " + + required.length + " verifier commands (incomplete)"); } - return { satisfied: reasons.length === 0, reasons }; + if (required.length > 0) { + const requiredSet = new Set(required); + for (const command of reported) { + if (!requiredSet.has(command)) reasons.push("verifier result for unrelated command `" + command + "`"); + } + } + return { satisfied: reasons.length === 0, reasons, missing }; +} + +/** One preflight row per distinct verifier command (#277). */ +export interface VerifierPreflightEntry { + command: string; + declared: boolean; + needsIsolation: boolean; + reason: string; +} + +/** + * Pure preflight over declared verifier effects (#277). It never runs a command + * or reads the filesystem; it only tells the caller which verifiers must not run + * on a shared checkout without isolation or main's confirmation. + */ +export function verifierPreflight(packet: DispatchPacket): VerifierPreflightEntry[] { + const effects = new Map((packet.verifierEffects ?? []).map((effect) => [effect.command.trim(), effect])); + return distinctCommands(packet.verifierCommands).map((command) => { + const effect = effects.get(command); + const declared = effect !== undefined; + if (packet.worktreePolicy === "isolated-write") { + return { command, declared, needsIsolation: false, reason: "packet is isolated-write" }; + } + if (!effect) { + return { command, declared, needsIsolation: true, + reason: "no declared write boundary on a shared-read packet; run it in an isolated copy or confirm with main" }; + } + if (effect.runInIsolation === true) { + return { command, declared, needsIsolation: true, reason: "declared runInIsolation" }; + } + if (effect.expectedWrites.length > 0) { + return { command, declared, needsIsolation: true, + reason: "declares writes (" + effect.expectedWrites.join(", ") + ") on a shared-read packet" }; + } + return { command, declared, needsIsolation: false, reason: "declared read-only (expectedWrites: [])" }; + }); } diff --git a/plugins/codexclaw/components/subagent-config/test/dispatch-contract.test.ts b/plugins/codexclaw/components/subagent-config/test/dispatch-contract.test.ts index ef144035..74c3362f 100644 --- a/plugins/codexclaw/components/subagent-config/test/dispatch-contract.test.ts +++ b/plugins/codexclaw/components/subagent-config/test/dispatch-contract.test.ts @@ -7,6 +7,7 @@ import { validatePacket, validateReceipt, receiptSatisfiesPacket, + verifierPreflight, type DispatchPacket, type DispatchReceipt, } from "../src/dispatch-contract.ts"; @@ -95,6 +96,7 @@ test("receiptSatisfiesPacket: matching pair is satisfied", () => { const result = receiptSatisfiesPacket(makePacket(), makeReceipt()); assert.equal(result.satisfied, true); assert.deepEqual(result.reasons, []); + assert.deepEqual(result.missing, []); }); test("receiptSatisfiesPacket: packetId mismatch", () => { @@ -140,3 +142,172 @@ test('architect packet roundtrips with main judgment ownership', () => { assert.deepEqual(validatePacket(JSON.parse(JSON.stringify(packet))), []); assert.ok(validatePacket({ ...packet, judgmentOwnership: 'architect' }).includes('judgmentOwnership must be main')); }); + + +// #276 / #277 (issue train 0930, devlog/_plan/260930_issue_train/010) + +function twoCommandPacket(overrides: Partial = {}): DispatchPacket { + return makePacket({ verifierCommands: ["first-check", "second-check"], ...overrides }); +} + +test("receiptSatisfiesPacket: #276 repro — unrelated single result for two required commands", () => { + const result = receiptSatisfiesPacket( + twoCommandPacket(), + makeReceipt({ commandsRun: ["unrelated-check"], verifierResult: { command: "unrelated-check", exitCode: 0, output: "ok" } }), + ); + assert.equal(result.satisfied, false); + assert.ok(result.reasons.some(r => r.includes("unrelated command"))); + assert.ok(result.reasons.some(r => r.includes("incomplete"))); + assert.deepEqual(result.missing, ["first-check", "second-check"]); +}); + +test("receiptSatisfiesPacket: single mismatched command is rejected", () => { + const result = receiptSatisfiesPacket( + makePacket({ verifierCommands: ["npm test"] }), + makeReceipt({ verifierResult: { command: "npm run test", exitCode: 0, output: "ok" } }), + ); + assert.equal(result.satisfied, false); + assert.deepEqual(result.missing, ["npm test"]); +}); + +test("receiptSatisfiesPacket: two required commands with one matching verifierResults entry", () => { + const result = receiptSatisfiesPacket( + twoCommandPacket(), + makeReceipt({ verifierResult: undefined, verifierResults: [{ command: "first-check", exitCode: 0, output: "ok" }] }), + ); + assert.equal(result.satisfied, false); + assert.deepEqual(result.missing, ["second-check"]); + assert.ok(result.reasons.some(r => r.includes("missing verifier result"))); +}); + +test("receiptSatisfiesPacket: legacy single matching result with two required commands is incomplete", () => { + const result = receiptSatisfiesPacket( + twoCommandPacket(), + makeReceipt({ verifierResult: { command: "first-check", exitCode: 0, output: "ok" } }), + ); + assert.equal(result.satisfied, false); + assert.ok(result.reasons.some(r => r.includes("legacy verifierResult") && r.includes("incomplete"))); +}); + +test("receiptSatisfiesPacket: verifierResults covering every command is satisfied", () => { + const result = receiptSatisfiesPacket( + twoCommandPacket(), + makeReceipt({ verifierResult: undefined, verifierResults: [ + { command: "first-check", exitCode: 0, output: "ok" }, + { command: "second-check", exitCode: 0, output: "ok" }, + ] }), + ); + assert.equal(result.satisfied, true); + assert.deepEqual(result.reasons, []); + assert.deepEqual(result.missing, []); +}); + +test("receiptSatisfiesPacket: unrelated extra result fails even with full coverage", () => { + const result = receiptSatisfiesPacket( + twoCommandPacket(), + makeReceipt({ verifierResult: undefined, verifierResults: [ + { command: "first-check", exitCode: 0, output: "ok" }, + { command: "second-check", exitCode: 0, output: "ok" }, + { command: "extra-check", exitCode: 0, output: "ok" }, + ] }), + ); + assert.equal(result.satisfied, false); + assert.ok(result.reasons.some(r => r.includes("unrelated command") && r.includes("extra-check"))); +}); + +test("receiptSatisfiesPacket: duplicate and padded packet commands match trimmed results", () => { + const result = receiptSatisfiesPacket( + makePacket({ verifierCommands: [" npm test ", "npm test"] }), + makeReceipt({ verifierResult: { command: "npm test", exitCode: 0, output: "ok" } }), + ); + assert.equal(result.satisfied, true); + assert.deepEqual(result.missing, []); +}); + +test("receiptSatisfiesPacket: nonzero matching result in verifierResults fails", () => { + const result = receiptSatisfiesPacket( + twoCommandPacket(), + makeReceipt({ verifierResult: undefined, verifierResults: [ + { command: "first-check", exitCode: 0, output: "ok" }, + { command: "second-check", exitCode: 2, output: "boom" }, + ] }), + ); + assert.equal(result.satisfied, false); + assert.ok(result.reasons.some(r => r.includes("exit code 2"))); +}); + +test("receiptSatisfiesPacket: malformed unvalidated result does not throw", () => { + for (const verifierResults of [[{ command: 1 }], {}]) { + const result = receiptSatisfiesPacket( + makePacket(), + makeReceipt({ verifierResult: undefined, verifierResults: verifierResults as any }), + ); + assert.equal(result.satisfied, false); + assert.ok(result.reasons.some(r => r.includes("malformed verifier results"))); + } +}); + +test("validateReceipt: rejects malformed verifier results", () => { + const legacy = makeReceipt({ verifierResult: { command: "npm test", exitCode: "0" as any, output: "" } }); + assert.ok(validateReceipt(legacy).some(e => e.startsWith("verifierResult must be"))); + const notArray = makeReceipt({ verifierResults: {} as any }); + assert.ok(validateReceipt(notArray).some(e => e.startsWith("verifierResults must be"))); + const badEntry = makeReceipt({ verifierResults: [{ command: 1 }] as any }); + assert.ok(validateReceipt(badEntry).some(e => e.startsWith("verifierResults must be"))); + assert.deepEqual(validateReceipt(makeReceipt({ verifierResults: [{ command: "npm test", exitCode: 0, output: "" }] })), []); +}); + +test("validatePacket: rejects blank verifier command entries", () => { + const errors = validatePacket(makePacket({ verifierCommands: ["npm test", " "] })); + assert.ok(errors.includes("verifierCommands entries must be non-empty strings")); +}); + +test("validatePacket: verifierEffects shape", () => { + const unknown = validatePacket(makePacket({ verifierEffects: [{ command: "other", expectedWrites: [] }] })); + assert.ok(unknown.some(e => e.includes("is not in verifierCommands"))); + const duplicate = validatePacket(makePacket({ verifierEffects: [ + { command: "npm test", expectedWrites: [] }, { command: " npm test", expectedWrites: [] }, + ] })); + assert.ok(duplicate.some(e => e.includes("more than once"))); + const writes = validatePacket(makePacket({ verifierEffects: [{ command: "npm test", expectedWrites: "x" as any }] })); + assert.ok(writes.some(e => e.includes("expectedWrites"))); + const isolation = validatePacket(makePacket({ verifierEffects: [{ command: "npm test", expectedWrites: [], runInIsolation: "yes" as any }] })); + assert.ok(isolation.some(e => e.includes("runInIsolation"))); + assert.deepEqual(validatePacket(makePacket({ verifierEffects: [{ command: "npm test", expectedWrites: [] }] })), []); +}); + +test("validatePacket: verifierEffects container and entry guards", () => { + assert.ok(validatePacket(makePacket({ verifierEffects: {} as any })).includes("verifierEffects must be an array")); + assert.ok(validatePacket(makePacket({ verifierEffects: [null] as any })).includes("verifierEffects entries must be objects")); + assert.ok(validatePacket(makePacket({ verifierEffects: [{ command: " ", expectedWrites: [] }] })) + .includes("verifierEffects command must be a non-empty string")); +}); + +test("verifierPreflight: shared-read flags undeclared and writing verifiers", () => { + const rows = verifierPreflight(makePacket({ + worktreePolicy: "shared-read", + verifierCommands: ["undeclared", "read-only", "writer", "isolated"], + verifierEffects: [ + { command: "read-only", expectedWrites: [] }, + { command: "writer", expectedWrites: ["cache.db"] }, + { command: "isolated", expectedWrites: [], runInIsolation: true }, + ], + })); + assert.deepEqual(rows.map(r => [r.command, r.declared, r.needsIsolation]), [ + ["undeclared", false, true], + ["read-only", true, false], + ["writer", true, true], + ["isolated", true, true], + ]); + assert.ok(rows[2].reason.includes("cache.db")); +}); + +test("verifierPreflight: isolated-write accepts declared writes", () => { + const rows = verifierPreflight(makePacket({ + worktreePolicy: "isolated-write", + verifierCommands: ["writer"], + verifierEffects: [{ command: "writer", expectedWrites: ["out/"] }], + })); + assert.deepEqual(rows, [{ command: "writer", declared: true, needsIsolation: false, reason: "packet is isolated-write" }]); +}); + diff --git a/plugins/codexclaw/skills/pabcd/references/delegation.md b/plugins/codexclaw/skills/pabcd/references/delegation.md index c0fad64e..67fcf4fa 100644 --- a/plugins/codexclaw/skills/pabcd/references/delegation.md +++ b/plugins/codexclaw/skills/pabcd/references/delegation.md @@ -22,6 +22,15 @@ Subagents return evidence and unresolved judgments; the main session decides and integrates. Dispatch only specifiable work whose coordination cost is justified (DISPATCH-ECONOMY-01). +**DISPATCH-VERIFIER-01 (DEFAULT).** When a packet names verifier commands, the +receipt reports one result per command; extra checks belong in the commands-run +list, not in the verifier results. A typed receipt satisfies its packet only when +every required command has a matching result with exit 0 and no result names a +command the packet did not require. Under a shared-read packet, declare each +verifier's writes (`expectedWrites: []` for read-only) or run it in an isolated +copy; an undeclared verifier goes back to main before it runs in a shared tree. +A declaration is the author's claim, not proof: codexclaw never executes it. + ### Optional worker progress checkpoint (#265) For a long bounded write packet, the coordinator may grant a specific `PROGRESS.md` diff --git a/structure/20_pabcd_dispatch_doctrine.md b/structure/20_pabcd_dispatch_doctrine.md index a3d513b8..31b9e444 100644 --- a/structure/20_pabcd_dispatch_doctrine.md +++ b/structure/20_pabcd_dispatch_doctrine.md @@ -251,6 +251,13 @@ codexclaw translation: drip-feed spawning taxes the main session's context once per return and fragments triage. This extends the "fan out before waiting" lifecycle rule from mechanics to economy. +- **DISPATCH-VERIFIER-01 (verifier coverage and effects).** A receipt reports one + result per packet verifier command, and a shared-read packet declares each + verifier's writes or runs it in isolation. E2 library contract: + `receiptSatisfiesPacket` and `verifierPreflight` in + `plugins/codexclaw/components/subagent-config/src/dispatch-contract.ts` (#276, + #277); E7 guidance in `plugins/codexclaw/skills/pabcd/references/delegation.md`. + No hook calls either function, and a declaration is a claim, not proof. --- diff --git a/structure/INDEX.md b/structure/INDEX.md index f5dd795e..36386565 100644 --- a/structure/INDEX.md +++ b/structure/INDEX.md @@ -137,7 +137,7 @@ the adapter preamble applies on load and `cxc-dev` discipline wins on conflict. ### `components/subagent-config` -Per-role subagent model, reasoning-effort, and prompt configuration. `src/store.ts` reads/writes `.codexclaw/subagents.json` atomically for `explorer`, `reviewer`, `executor`, and `architect`, defaulting each role to the main Codex model with inherited effort (`effort: null`; valid overrides are the catalog-supported values low/medium/high/xhigh). `src/catalog.ts` builds a selectable model catalog from the native Codex cache allowlist plus optional ocx-backed model ids, with native models first. `src/mcp.ts` serves a stdio MCP server with `subagents_get`, `subagents_set`, and `catalog_list` tools. +Per-role subagent model, reasoning-effort, and prompt configuration. `src/store.ts` reads/writes `.codexclaw/subagents.json` atomically for `explorer`, `reviewer`, `executor`, and `architect`, defaulting each role to the main Codex model with inherited effort (`effort: null`; valid overrides are the catalog-supported values low/medium/high/xhigh). `src/catalog.ts` builds a selectable model catalog from the native Codex cache allowlist plus optional ocx-backed model ids, with native models first. `src/mcp.ts` serves a stdio MCP server with `subagents_get`, `subagents_set`, and `catalog_list` tools. `src/dispatch-contract.ts` holds the typed DispatchPacket/DispatchReceipt contract (#17): verifier coverage for receipts (#276) and a pure verifier-effects preflight (#277); no runtime path calls it yet. --- From 5d14db1b8b2380e135ae1864356e7a872d68fbfe Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Wed, 30 Sep 2026 02:26:16 +0900 Subject: [PATCH 79/90] fix(dispatch): fail closed on malformed commands and falsy legacy results; align DISPATCH-VERIFIER-01 prose; +6 tests, badges 3729 --- README.ko.md | 2 +- README.md | 2 +- README.zh.md | 2 +- .../010_wp2_dispatch_contract.md | 17 +++++- .../subagent-config/dist/dispatch-contract.js | 34 ++++++++---- .../subagent-config/src/dispatch-contract.ts | 34 ++++++++---- .../test/dispatch-contract.test.ts | 53 +++++++++++++++++++ .../skills/pabcd/references/delegation.md | 11 ++-- structure/20_pabcd_dispatch_doctrine.md | 5 +- 9 files changed, 130 insertions(+), 30 deletions(-) diff --git a/README.ko.md b/README.ko.md index d2965583..dfa8c773 100644 --- a/README.ko.md +++ b/README.ko.md @@ -13,7 +13,7 @@

CI - 3,708 tests + 3,729 tests 29 skills 31 hooks Documentation diff --git a/README.md b/README.md index 275275d8..cb6343d8 100644 --- a/README.md +++ b/README.md @@ -13,7 +13,7 @@

CI - 3,708 tests + 3,729 tests 29 skills 31 hooks Documentation diff --git a/README.zh.md b/README.zh.md index 0d721a8f..5b071503 100644 --- a/README.zh.md +++ b/README.zh.md @@ -13,7 +13,7 @@

CI - 3,708 tests + 3,729 tests 29 skills 31 hooks Documentation diff --git a/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md b/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md index d56b086c..0b2620fd 100644 --- a/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md +++ b/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md @@ -262,7 +262,7 @@ Import `verifierPreflight` beside the existing imports (`:6-12`). Existing tests | `verifierPreflight: shared-read flags undeclared and writing verifiers` | shared-read with undeclared, `[]`, writes, `runInIsolation` | `needsIsolation` true/false/true/true, `declared` false/true/true/true | | `verifierPreflight: isolated-write accepts declared writes` | isolated-write with writes | `needsIsolation:false`, `declared:true` | -Fifteen new tests. The README test badges move by the measured delta (040). +Fifteen new tests planned; C added six more (see "wp2 C record"). The README test badges and `inventory.json` move by the measured delta in this phase (`000_plan.md` file map: badges each phase). ### 4. MODIFY `plugins/codexclaw/skills/pabcd/references/delegation.md` @@ -289,7 +289,7 @@ These are this phase's SOT-SYNC-01 targets. ## Scope boundary -IN: the six files above (including `structure/20_pabcd_dispatch_doctrine.md`). OUT: `commandsRun` semantics, `sourceIdentity`, release-gate's `dispatch-contracts` receipt (`pabcd-state/src/release-gate.ts:393`, stays missing), #256/#273 receipt fields, any hook. +IN: the six files above (including `structure/20_pabcd_dispatch_doctrine.md`), plus the test badges in `README.md`, `README.ko.md`, `README.zh.md` and `inventory.json` via `inventory.mjs --write --tests `. OUT: `commandsRun` semantics, `sourceIdentity`, release-gate's `dispatch-contracts` receipt (`pabcd-state/src/release-gate.ts:393`, stays missing), #256/#273 receipt fields, any hook. ## Verification (PLAN-VERIFIER-REAL-01) @@ -315,3 +315,16 @@ Red-green: the #276 repro test and the mismatched-command test must fail against Continuity (LOOP-CONTINUITY-01), quoting the wp1 D summary in 000: "the roadmap is locked and wp2 builds 010 as written"; its negative side (context-only review independence, no stated resource bound for wp4, the c-10 correction) is carried forward. This P keeps that direction. Re-checked on `codex/issue-train-0930-wp2`: `git diff --stat 659de59b..HEAD -- plugins structure` is empty, so every anchor above still holds. One amendment: the `structure/INDEX.md` subagent-config section (`:138-140`) is a prose paragraph, so §5 appends one sentence to it: "`src/dispatch-contract.ts` holds the typed DispatchPacket/DispatchReceipt contract (#17): verifier coverage for receipts (#276) and a pure verifier-effects preflight (#277); no runtime path calls it yet." The `structure/20` bullet goes after the DISPATCH-ECONOMY-01 bullet's last clause (`:253`, "from mechanics to economy."), before the `---` that closes §3. Architect consultation: after the reflection at `048a521c`, A folds added a fail-closed malformed-result branch to (1g), including the non-array case; it implements D4 (every result must be a passing, matching result) and D8 (both result fields are shape-checked), so it changes no design decision. This amendment changes placement only, so the phase-audit architect-recheck trigger is not met. + +## wp2 C record (2026-09-30) + +C review round 1 (fresh implementation reviewer `01a0ee2f-7b70`, initiative verifier `01a0ee2f-7cbd`, in parallel) on `b852eb24`: GO-WITH-FIXES (1) and GO-WITH-FIXES (4). Accepted and fixed: + +- The §4 prose said "declare writes or isolate", but under shared-read the code (correctly, D13) isolates any verifier that declares writes. DISPATCH-VERIFIER-01 in `delegation.md` and `structure/20` now say only a verifier declared read-only runs in the shared tree; the unrelated-result clause and the function docstring carry D5's "when the packet requires commands". +- A present but falsy legacy `verifierResult` (`null`) slipped past the malformed check; presence now tests `!== undefined`. +- An unvalidated packet with a non-string command made both functions throw; both now fail closed (`packet has malformed verifier commands`; preflight emits a `needsIsolation: true` row for it). +- Untriggered paths got six tests: legacy plus array merge (D2), zero required commands with a stray passing result (D5), no results suppressing per-command reasons, falsy legacy result, non-string packet commands, element-level `expectedWrites` guard. Total new tests: 21. +- Badges move in this phase (000's file map), so 010's IN list gained the READMEs and `inventory.json`. +- Red record. Method: `/tmp/it0930/wp2-red.sh` copies the component, restores `src/dispatch-contract.ts` from `659de59b`, stubs `verifierPreflight` as absent, and runs the committed test file with `node --test`. Result: 38 tests, 18 pass, 20 fail, including the #276 repro and the mismatched-command test. On the new source the same file passes 38/38. The earlier "14 failing" in the B->C attest came from an intermediate test file (30 tests, preflight tests removed) and is superseded by this record. + +Disclosure for the 040 CHANGELOG: besides the command-match change, `validateReceipt` now rejects a legacy `verifierResult` without a string `output` or an integer `exitCode`, and `validatePacket` rejects blank or non-string `verifierCommands` entries; #277's "existing packets remain readable" holds for well-formed packets. diff --git a/plugins/codexclaw/components/subagent-config/dist/dispatch-contract.js b/plugins/codexclaw/components/subagent-config/dist/dispatch-contract.js index a779251a..75a1c4a2 100644 --- a/plugins/codexclaw/components/subagent-config/dist/dispatch-contract.js +++ b/plugins/codexclaw/components/subagent-config/dist/dispatch-contract.js @@ -94,9 +94,15 @@ function isVerifierResult(value ) { return typeof v.command === "string" && Number.isInteger(v.exitCode) && typeof v.output === "string"; } -/** Distinct trimmed commands; exact string equality after trim, no other normalization. */ -function distinctCommands(commands ) { - return [...new Set(commands.map((command) => command.trim()))]; +/** Distinct trimmed string commands; exact equality after trim, no other normalization. */ +function distinctCommands(commands ) { + return [...new Set(commands.filter((command) => typeof command === "string") + .map((command) => command.trim()))]; +} + +/** Entries an unvalidated packet carries that are not non-blank strings. */ +function malformedCommandCount(commands ) { + return commands.filter((command) => typeof command !== "string" || !command.trim()).length; } function validateVerifierEffects(value , commands ) { @@ -191,7 +197,8 @@ export function validateReceipt(receipt ) { /** * Check that a receipt satisfies its packet's verifier requirements (#276). * Every distinct required command needs a matching result, every result must - * exit 0, and a result for a command the packet did not require is rejected. + * exit 0, and when the packet requires commands, a result for a command it did + * not require is rejected. */ export function receiptSatisfiesPacket(packet , receipt ) @@ -206,10 +213,14 @@ export function receiptSatisfiesPacket(packet , receipt if (receipt.status !== "complete") { reasons.push("receipt status is " + receipt.status + ", not complete"); } - const required = distinctCommands(packet.verifierCommands); + const commands = Array.isArray(packet.verifierCommands) ? packet.verifierCommands : []; + if (!Array.isArray(packet.verifierCommands) || malformedCommandCount(commands) > 0) { + reasons.push("packet has malformed verifier commands (see validatePacket)"); + } + const required = distinctCommands(commands).filter((command) => command.length > 0); const reportedResults = [ ...(Array.isArray(receipt.verifierResults) ? receipt.verifierResults : []), - ...(receipt.verifierResult ? [receipt.verifierResult] : []), + ...(receipt.verifierResult !== undefined ? [receipt.verifierResult] : []), ]; // Unvalidated input must not throw here; malformed entries fail the receipt. const results = reportedResults.filter(isVerifierResult); @@ -230,7 +241,7 @@ export function receiptSatisfiesPacket(packet , receipt if (results.length > 0) { for (const command of missing) reasons.push("missing verifier result for `" + command + "`"); } - if (required.length > 1 && receipt.verifierResults === undefined && receipt.verifierResult) { + if (required.length > 1 && receipt.verifierResults === undefined && receipt.verifierResult !== undefined) { reasons.push("receipt reports one legacy verifierResult; packet requires " + required.length + " verifier commands (incomplete)"); } @@ -258,7 +269,12 @@ export function receiptSatisfiesPacket(packet , receipt */ export function verifierPreflight(packet ) { const effects = new Map((packet.verifierEffects ?? []).map((effect) => [effect.command.trim(), effect])); - return distinctCommands(packet.verifierCommands).map((command) => { + const commands = Array.isArray(packet.verifierCommands) ? packet.verifierCommands : []; + const malformed = commands + .filter((command) => typeof command !== "string" || !command.trim()) + .map((command) => ({ command: String(command), declared: false, needsIsolation: true, + reason: "malformed verifier command (see validatePacket); do not run it" })); + return [...distinctCommands(commands).filter((command) => command.length > 0).map((command) => { const effect = effects.get(command); const declared = effect !== undefined; if (packet.worktreePolicy === "isolated-write") { @@ -276,6 +292,6 @@ export function verifierPreflight(packet ) reason: "declares writes (" + effect.expectedWrites.join(", ") + ") on a shared-read packet" }; } return { command, declared, needsIsolation: false, reason: "declared read-only (expectedWrites: [])" }; - }); + }), ...malformed]; } diff --git a/plugins/codexclaw/components/subagent-config/src/dispatch-contract.ts b/plugins/codexclaw/components/subagent-config/src/dispatch-contract.ts index 6fccc893..a9816c94 100644 --- a/plugins/codexclaw/components/subagent-config/src/dispatch-contract.ts +++ b/plugins/codexclaw/components/subagent-config/src/dispatch-contract.ts @@ -94,9 +94,15 @@ function isVerifierResult(value: unknown): value is VerifierResult { return typeof v.command === "string" && Number.isInteger(v.exitCode) && typeof v.output === "string"; } -/** Distinct trimmed commands; exact string equality after trim, no other normalization. */ -function distinctCommands(commands: readonly string[]): string[] { - return [...new Set(commands.map((command) => command.trim()))]; +/** Distinct trimmed string commands; exact equality after trim, no other normalization. */ +function distinctCommands(commands: readonly unknown[]): string[] { + return [...new Set(commands.filter((command): command is string => typeof command === "string") + .map((command) => command.trim()))]; +} + +/** Entries an unvalidated packet carries that are not non-blank strings. */ +function malformedCommandCount(commands: readonly unknown[]): number { + return commands.filter((command) => typeof command !== "string" || !command.trim()).length; } function validateVerifierEffects(value: unknown, commands: unknown): string[] { @@ -191,7 +197,8 @@ export function validateReceipt(receipt: unknown): string[] { /** * Check that a receipt satisfies its packet's verifier requirements (#276). * Every distinct required command needs a matching result, every result must - * exit 0, and a result for a command the packet did not require is rejected. + * exit 0, and when the packet requires commands, a result for a command it did + * not require is rejected. */ export function receiptSatisfiesPacket(packet: DispatchPacket, receipt: DispatchReceipt): { satisfied: boolean; @@ -206,10 +213,14 @@ export function receiptSatisfiesPacket(packet: DispatchPacket, receipt: Dispatch if (receipt.status !== "complete") { reasons.push("receipt status is " + receipt.status + ", not complete"); } - const required = distinctCommands(packet.verifierCommands); + const commands: unknown[] = Array.isArray(packet.verifierCommands) ? packet.verifierCommands : []; + if (!Array.isArray(packet.verifierCommands) || malformedCommandCount(commands) > 0) { + reasons.push("packet has malformed verifier commands (see validatePacket)"); + } + const required = distinctCommands(commands).filter((command) => command.length > 0); const reportedResults: unknown[] = [ ...(Array.isArray(receipt.verifierResults) ? receipt.verifierResults : []), - ...(receipt.verifierResult ? [receipt.verifierResult] : []), + ...(receipt.verifierResult !== undefined ? [receipt.verifierResult] : []), ]; // Unvalidated input must not throw here; malformed entries fail the receipt. const results = reportedResults.filter(isVerifierResult); @@ -230,7 +241,7 @@ export function receiptSatisfiesPacket(packet: DispatchPacket, receipt: Dispatch if (results.length > 0) { for (const command of missing) reasons.push("missing verifier result for `" + command + "`"); } - if (required.length > 1 && receipt.verifierResults === undefined && receipt.verifierResult) { + if (required.length > 1 && receipt.verifierResults === undefined && receipt.verifierResult !== undefined) { reasons.push("receipt reports one legacy verifierResult; packet requires " + required.length + " verifier commands (incomplete)"); } @@ -258,7 +269,12 @@ export interface VerifierPreflightEntry { */ export function verifierPreflight(packet: DispatchPacket): VerifierPreflightEntry[] { const effects = new Map((packet.verifierEffects ?? []).map((effect) => [effect.command.trim(), effect])); - return distinctCommands(packet.verifierCommands).map((command) => { + const commands: unknown[] = Array.isArray(packet.verifierCommands) ? packet.verifierCommands : []; + const malformed: VerifierPreflightEntry[] = commands + .filter((command) => typeof command !== "string" || !command.trim()) + .map((command) => ({ command: String(command), declared: false, needsIsolation: true, + reason: "malformed verifier command (see validatePacket); do not run it" })); + return [...distinctCommands(commands).filter((command) => command.length > 0).map((command) => { const effect = effects.get(command); const declared = effect !== undefined; if (packet.worktreePolicy === "isolated-write") { @@ -276,6 +292,6 @@ export function verifierPreflight(packet: DispatchPacket): VerifierPreflightEntr reason: "declares writes (" + effect.expectedWrites.join(", ") + ") on a shared-read packet" }; } return { command, declared, needsIsolation: false, reason: "declared read-only (expectedWrites: [])" }; - }); + }), ...malformed]; } diff --git a/plugins/codexclaw/components/subagent-config/test/dispatch-contract.test.ts b/plugins/codexclaw/components/subagent-config/test/dispatch-contract.test.ts index 74c3362f..3c4abaab 100644 --- a/plugins/codexclaw/components/subagent-config/test/dispatch-contract.test.ts +++ b/plugins/codexclaw/components/subagent-config/test/dispatch-contract.test.ts @@ -311,3 +311,56 @@ test("verifierPreflight: isolated-write accepts declared writes", () => { assert.deepEqual(rows, [{ command: "writer", declared: true, needsIsolation: false, reason: "packet is isolated-write" }]); }); + +test("receiptSatisfiesPacket: legacy verifierResult merges with verifierResults", () => { + const result = receiptSatisfiesPacket( + twoCommandPacket(), + makeReceipt({ + verifierResults: [{ command: "first-check", exitCode: 0, output: "ok" }], + verifierResult: { command: "second-check", exitCode: 0, output: "ok" }, + }), + ); + assert.equal(result.satisfied, true); + assert.ok(!result.reasons.some(r => r.includes("legacy"))); +}); + +test("receiptSatisfiesPacket: no required commands accepts a stray passing result", () => { + const result = receiptSatisfiesPacket( + makePacket({ verifierCommands: [] }), + makeReceipt({ verifierResult: { command: "extra-check", exitCode: 0, output: "ok" } }), + ); + assert.equal(result.satisfied, true); +}); + +test("receiptSatisfiesPacket: no results reports one reason and lists every missing command", () => { + const result = receiptSatisfiesPacket(twoCommandPacket(), makeReceipt({ verifierResult: undefined })); + assert.equal(result.satisfied, false); + assert.deepEqual(result.missing, ["first-check", "second-check"]); + assert.ok(result.reasons.includes("packet has verifier commands but receipt has no verifier result")); + assert.ok(!result.reasons.some(r => r.startsWith("missing verifier result"))); +}); + +test("receiptSatisfiesPacket: a present but falsy legacy verifierResult is malformed", () => { + const result = receiptSatisfiesPacket( + makePacket(), + makeReceipt({ verifierResult: null as any, verifierResults: [{ command: "npm test", exitCode: 0, output: "ok" }] }), + ); + assert.equal(result.satisfied, false); + assert.ok(result.reasons.some(r => r.includes("malformed verifier results"))); +}); + +test("unvalidated packets with non-string commands fail closed without throwing", () => { + const packet = makePacket({ verifierCommands: ["npm test", 1 as any] }); + const result = receiptSatisfiesPacket(packet, makeReceipt()); + assert.equal(result.satisfied, false); + assert.ok(result.reasons.some(r => r.includes("malformed verifier commands"))); + const rows = verifierPreflight(packet); + assert.deepEqual(rows.map(r => [r.command, r.needsIsolation]), [["npm test", true], ["1", true]]); + assert.ok(rows[1].reason.includes("malformed verifier command")); +}); + +test("validatePacket: verifierEffects expectedWrites entries must be non-blank strings", () => { + const errors = validatePacket(makePacket({ verifierEffects: [{ command: "npm test", expectedWrites: ["out/", " "] }] })); + assert.ok(errors.some(e => e.includes("expectedWrites") && e.includes("non-empty strings"))); +}); + diff --git a/plugins/codexclaw/skills/pabcd/references/delegation.md b/plugins/codexclaw/skills/pabcd/references/delegation.md index 67fcf4fa..c88a78ce 100644 --- a/plugins/codexclaw/skills/pabcd/references/delegation.md +++ b/plugins/codexclaw/skills/pabcd/references/delegation.md @@ -25,11 +25,12 @@ integrates. Dispatch only specifiable work whose coordination cost is justified **DISPATCH-VERIFIER-01 (DEFAULT).** When a packet names verifier commands, the receipt reports one result per command; extra checks belong in the commands-run list, not in the verifier results. A typed receipt satisfies its packet only when -every required command has a matching result with exit 0 and no result names a -command the packet did not require. Under a shared-read packet, declare each -verifier's writes (`expectedWrites: []` for read-only) or run it in an isolated -copy; an undeclared verifier goes back to main before it runs in a shared tree. -A declaration is the author's claim, not proof: codexclaw never executes it. +every required command has a matching result with exit 0 and, when the packet +requires commands, no result names a command it did not require. Under a +shared-read packet, only a verifier declared read-only (`expectedWrites: []`) +runs in the shared tree; one that declares writes, asks for isolation or declares +nothing runs in an isolated copy or goes back to main first. A declaration is the +author's claim, not proof: codexclaw never executes it. ### Optional worker progress checkpoint (#265) diff --git a/structure/20_pabcd_dispatch_doctrine.md b/structure/20_pabcd_dispatch_doctrine.md index 31b9e444..af9a5f7a 100644 --- a/structure/20_pabcd_dispatch_doctrine.md +++ b/structure/20_pabcd_dispatch_doctrine.md @@ -252,8 +252,9 @@ codexclaw translation: fragments triage. This extends the "fan out before waiting" lifecycle rule from mechanics to economy. - **DISPATCH-VERIFIER-01 (verifier coverage and effects).** A receipt reports one - result per packet verifier command, and a shared-read packet declares each - verifier's writes or runs it in isolation. E2 library contract: + result per packet verifier command. Under a shared-read packet, only a verifier + declared read-only runs in the shared tree; any other runs isolated or returns + to main. E2 library contract: `receiptSatisfiesPacket` and `verifierPreflight` in `plugins/codexclaw/components/subagent-config/src/dispatch-contract.ts` (#276, #277); E7 guidance in `plugins/codexclaw/skills/pabcd/references/delegation.md`. From ef1ce80df89678aee8a0e1a3f10830211e977547 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Wed, 30 Sep 2026 02:29:57 +0900 Subject: [PATCH 80/90] fix(dispatch): preflight ignores malformed verifier effects (fail closed); badges 3730 --- README.ko.md | 2 +- README.md | 2 +- README.zh.md | 2 +- .../_plan/260930_issue_train/010_wp2_dispatch_contract.md | 2 +- .../components/subagent-config/dist/dispatch-contract.js | 6 +++++- .../components/subagent-config/src/dispatch-contract.ts | 6 +++++- .../subagent-config/test/dispatch-contract.test.ts | 7 +++++++ 7 files changed, 21 insertions(+), 6 deletions(-) diff --git a/README.ko.md b/README.ko.md index dfa8c773..07fb562a 100644 --- a/README.ko.md +++ b/README.ko.md @@ -13,7 +13,7 @@

CI - 3,729 tests + 3,730 tests 29 skills 31 hooks Documentation diff --git a/README.md b/README.md index cb6343d8..5a8095c9 100644 --- a/README.md +++ b/README.md @@ -13,7 +13,7 @@

CI - 3,729 tests + 3,730 tests 29 skills 31 hooks Documentation diff --git a/README.zh.md b/README.zh.md index 5b071503..426029e9 100644 --- a/README.zh.md +++ b/README.zh.md @@ -13,7 +13,7 @@

CI - 3,729 tests + 3,730 tests 29 skills 31 hooks Documentation diff --git a/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md b/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md index 0b2620fd..89c023e7 100644 --- a/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md +++ b/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md @@ -323,7 +323,7 @@ C review round 1 (fresh implementation reviewer `01a0ee2f-7b70`, initiative veri - The §4 prose said "declare writes or isolate", but under shared-read the code (correctly, D13) isolates any verifier that declares writes. DISPATCH-VERIFIER-01 in `delegation.md` and `structure/20` now say only a verifier declared read-only runs in the shared tree; the unrelated-result clause and the function docstring carry D5's "when the packet requires commands". - A present but falsy legacy `verifierResult` (`null`) slipped past the malformed check; presence now tests `!== undefined`. - An unvalidated packet with a non-string command made both functions throw; both now fail closed (`packet has malformed verifier commands`; preflight emits a `needsIsolation: true` row for it). -- Untriggered paths got six tests: legacy plus array merge (D2), zero required commands with a stray passing result (D5), no results suppressing per-command reasons, falsy legacy result, non-string packet commands, element-level `expectedWrites` guard. Total new tests: 21. +- Untriggered paths got six tests: legacy plus array merge (D2), zero required commands with a stray passing result (D5), no results suppressing per-command reasons, falsy legacy result, non-string packet commands, element-level `expectedWrites` guard. Total new tests: 21, plus one in C round 2 for malformed `verifierEffects` in preflight (22; badges 3730). - Badges move in this phase (000's file map), so 010's IN list gained the READMEs and `inventory.json`. - Red record. Method: `/tmp/it0930/wp2-red.sh` copies the component, restores `src/dispatch-contract.ts` from `659de59b`, stubs `verifierPreflight` as absent, and runs the committed test file with `node --test`. Result: 38 tests, 18 pass, 20 fail, including the #276 repro and the mismatched-command test. On the new source the same file passes 38/38. The earlier "14 failing" in the B->C attest came from an intermediate test file (30 tests, preflight tests removed) and is superseded by this record. diff --git a/plugins/codexclaw/components/subagent-config/dist/dispatch-contract.js b/plugins/codexclaw/components/subagent-config/dist/dispatch-contract.js index 75a1c4a2..369ab384 100644 --- a/plugins/codexclaw/components/subagent-config/dist/dispatch-contract.js +++ b/plugins/codexclaw/components/subagent-config/dist/dispatch-contract.js @@ -268,7 +268,11 @@ export function receiptSatisfiesPacket(packet , receipt * on a shared checkout without isolation or main's confirmation. */ export function verifierPreflight(packet ) { - const effects = new Map((packet.verifierEffects ?? []).map((effect) => [effect.command.trim(), effect])); + // Malformed effect entries (unvalidated input) are ignored, so their command reads as undeclared. + const declaredEffects = (Array.isArray(packet.verifierEffects) ? packet.verifierEffects : []) + .filter((effect) => !!effect && typeof effect === "object" + && typeof effect.command === "string" && Array.isArray(effect.expectedWrites)); + const effects = new Map(declaredEffects.map((effect) => [effect.command.trim(), effect])); const commands = Array.isArray(packet.verifierCommands) ? packet.verifierCommands : []; const malformed = commands .filter((command) => typeof command !== "string" || !command.trim()) diff --git a/plugins/codexclaw/components/subagent-config/src/dispatch-contract.ts b/plugins/codexclaw/components/subagent-config/src/dispatch-contract.ts index a9816c94..90b0ba2f 100644 --- a/plugins/codexclaw/components/subagent-config/src/dispatch-contract.ts +++ b/plugins/codexclaw/components/subagent-config/src/dispatch-contract.ts @@ -268,7 +268,11 @@ export interface VerifierPreflightEntry { * on a shared checkout without isolation or main's confirmation. */ export function verifierPreflight(packet: DispatchPacket): VerifierPreflightEntry[] { - const effects = new Map((packet.verifierEffects ?? []).map((effect) => [effect.command.trim(), effect])); + // Malformed effect entries (unvalidated input) are ignored, so their command reads as undeclared. + const declaredEffects = (Array.isArray(packet.verifierEffects) ? packet.verifierEffects : []) + .filter((effect): effect is VerifierEffect => !!effect && typeof effect === "object" + && typeof effect.command === "string" && Array.isArray(effect.expectedWrites)); + const effects = new Map(declaredEffects.map((effect) => [effect.command.trim(), effect])); const commands: unknown[] = Array.isArray(packet.verifierCommands) ? packet.verifierCommands : []; const malformed: VerifierPreflightEntry[] = commands .filter((command) => typeof command !== "string" || !command.trim()) diff --git a/plugins/codexclaw/components/subagent-config/test/dispatch-contract.test.ts b/plugins/codexclaw/components/subagent-config/test/dispatch-contract.test.ts index 3c4abaab..bfbdaeab 100644 --- a/plugins/codexclaw/components/subagent-config/test/dispatch-contract.test.ts +++ b/plugins/codexclaw/components/subagent-config/test/dispatch-contract.test.ts @@ -364,3 +364,10 @@ test("validatePacket: verifierEffects expectedWrites entries must be non-blank s assert.ok(errors.some(e => e.includes("expectedWrites") && e.includes("non-empty strings"))); }); + +test("verifierPreflight: malformed effects are ignored and the command stays undeclared", () => { + for (const verifierEffects of [{}, [null], [{ command: 1, expectedWrites: [] }], [{ command: "npm test" }]]) { + const rows = verifierPreflight(makePacket({ verifierEffects: verifierEffects as any })); + assert.deepEqual(rows.map(r => [r.command, r.declared, r.needsIsolation]), [["npm test", false, true]]); + } +}); From e8fbe0961b0719c5864347f74a859f79bcba3399 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Wed, 30 Sep 2026 02:47:50 +0900 Subject: [PATCH 81/90] docs(plan): wp2 D summary; wp3 P revalidation --- .../_plan/260930_issue_train/010_wp2_dispatch_contract.md | 6 ++++++ .../260930_issue_train/020_wp3_interview_assumptions.md | 4 ++++ 2 files changed, 10 insertions(+) diff --git a/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md b/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md index 89c023e7..6a6852a4 100644 --- a/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md +++ b/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md @@ -328,3 +328,9 @@ C review round 1 (fresh implementation reviewer `01a0ee2f-7b70`, initiative veri - Red record. Method: `/tmp/it0930/wp2-red.sh` copies the component, restores `src/dispatch-contract.ts` from `659de59b`, stubs `verifierPreflight` as absent, and runs the committed test file with `node --test`. Result: 38 tests, 18 pass, 20 fail, including the #276 repro and the mismatched-command test. On the new source the same file passes 38/38. The earlier "14 failing" in the B->C attest came from an intermediate test file (30 tests, preflight tests removed) and is superseded by this record. Disclosure for the 040 CHANGELOG: besides the command-match change, `validateReceipt` now rejects a legacy `verifierResult` without a string `output` or an integer `exitCode`, and `validatePacket` rejects blank or non-string `verifierCommands` entries; #277's "existing packets remain readable" holds for well-formed packets. + +## wp2 D summary (2026-09-30) + +Conclusion: #276 and #277 are fixed and merged into `dev` through PR #278 (head `ef1ce80d`, 14/14 checks, merge `069a7d0e`). Evidence: the red check (20/38 failing on the `659de59b` source, including the #276 repro), 39/39 on the new source, and the C gate under `cxc receipt test` (3730 tests, 0 failures, inventory, gate, smoke, empty hook diff). Next: wp3 builds 020. + +What did not go well: the first build shipped prose that contradicted the preflight rule (declared writes still need isolation), and three conditional paths plus two malformed-input paths were untested until the C reviewers probed them; C took two review rounds and three gate runs. One full-suite run failed on an unrelated timing assertion (`spawn-attach-hook.test.ts:920`, 0.5 ms to 6.6 ms under concurrent load) that passed 3/3 in isolation; it is a latent flake worth watching in CI. Evidence that this direction is wrong: a caller appears that legitimately reports extra passing checks in `verifierResults` and is broken by D5. diff --git a/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md b/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md index 44c57af5..aaa39891 100644 --- a/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md +++ b/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md @@ -134,3 +134,7 @@ No conditional code path is added, so C-ACTIVATION-GROUNDING-01 does not apply; ## Enforcement naming (PLAN-BYPASS-NAMED-01) Tier E7 (agent-followed guidance). Executing surface: the main session writing the plan. Known bypass: an agent can still label an entry `user_confirmed` without a real `eventId`; nothing checks the reference against the ledger. Residual risk: acceptance check 1 holds by discipline plus reviewability (the reference is visible and checkable in the hashed plan), not by a gate. Wording: guidance, never "cannot"; the acceptance table above reads as "the rule requires". Final enforcement layer: none. + +## wp3 P revalidation (2026-09-30) + +Continuity (LOOP-CONTINUITY-01), quoting the wp2 D summary in 010: "#276 and #277 are fixed and merged ... Next: wp3 builds 020." This P keeps that direction. Re-checked on `codex/issue-train-0930-wp3` from `origin/dev` `069a7d0e`: `git diff --stat 659de59b..HEAD -- plugins/codexclaw/skills/interview plugins/codexclaw/skills/loop/references/durable-goalplan.md` is empty, and the quoted lines (`SKILL.md:28`, `:46-53`, `:170-172`, `:176-178`, `mind-dispatch.md:47-49`, `durable-goalplan.md:40`) read as planned. No amendment; the architect's D17-D21 stand, so no re-consultation. From 4254bc384b0d65928f108d02db10e8e3873e43e1 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Wed, 30 Sep 2026 02:49:24 +0900 Subject: [PATCH 82/90] docs(interview): assumption provenance at handoff, INTERVIEW-ASSUME-01 (#275) --- .../010_wp2_dispatch_contract.md | 2 +- .../020_wp3_interview_assumptions.md | 4 +- plugins/codexclaw/skills/interview/SKILL.md | 44 +++++++++++++++++-- .../interview/references/mind-dispatch.md | 5 ++- .../loop/references/durable-goalplan.md | 5 ++- 5 files changed, 52 insertions(+), 8 deletions(-) diff --git a/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md b/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md index 6a6852a4..8117c5f3 100644 --- a/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md +++ b/devlog/_plan/260930_issue_train/010_wp2_dispatch_contract.md @@ -331,6 +331,6 @@ Disclosure for the 040 CHANGELOG: besides the command-match change, `validateRec ## wp2 D summary (2026-09-30) -Conclusion: #276 and #277 are fixed and merged into `dev` through PR #278 (head `ef1ce80d`, 14/14 checks, merge `069a7d0e`). Evidence: the red check (20/38 failing on the `659de59b` source, including the #276 repro), 39/39 on the new source, and the C gate under `cxc receipt test` (3730 tests, 0 failures, inventory, gate, smoke, empty hook diff). Next: wp3 builds 020. +Conclusion: #276 and #277 are fixed and merged into `dev` through PR #278 (head `ef1ce80d`, 14/14 checks, merge `069a7d0e`). Evidence: the red check (20/38 failing on the `659de59b` source, including the #276 repro), 38/38 on the new source for the same file (39/39 after C round 2 added one test), and the C gate under `cxc receipt test` (3730 tests, 0 failures, inventory, gate, smoke, empty hook diff). Next: wp3 builds 020. What did not go well: the first build shipped prose that contradicted the preflight rule (declared writes still need isolation), and three conditional paths plus two malformed-input paths were untested until the C reviewers probed them; C took two review rounds and three gate runs. One full-suite run failed on an unrelated timing assertion (`spawn-attach-hook.test.ts:920`, 0.5 ms to 6.6 ms under concurrent load) that passed 3/3 in isolation; it is a latent flake worth watching in CI. Evidence that this direction is wrong: a caller appears that legitimately reports extra passing checks in `verifierResults` and is broken by D5. diff --git a/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md b/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md index aaa39891..20106f8f 100644 --- a/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md +++ b/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md @@ -115,7 +115,7 @@ SoT sync (SOT-SYNC-01): the Interview skill is the canonical owner of these rule | Check | Where it is met | |---|---| -| 1. inference cannot be presented as user-confirmed without an answer reference (rule-level; see enforcement naming) | 1(b) status bullet: confirmed/rejected require the answer `eventId` | +| 1. the rule requires an answer reference before an inference is presented as user-confirmed (rule-level; see enforcement naming) | 1(b) status bullet: confirmed/rejected require the answer `eventId` | | 2. rejected inference kept as decision trace, not carried as open | 1(b) last-but-one bullet: `## ASSUMPTION DECISIONS`, never in tracker assumptions | | 3. closeout distinguishes confirmed requirements from open inferred assumptions | 1(d), 1(e) | | 4. existing trackers and freeze manifests remain readable | no code change; 1(b) last bullet | @@ -138,3 +138,5 @@ Tier E7 (agent-followed guidance). Executing surface: the main session writing t ## wp3 P revalidation (2026-09-30) Continuity (LOOP-CONTINUITY-01), quoting the wp2 D summary in 010: "#276 and #277 are fixed and merged ... Next: wp3 builds 020." This P keeps that direction. Re-checked on `codex/issue-train-0930-wp3` from `origin/dev` `069a7d0e`: `git diff --stat 659de59b..HEAD -- plugins/codexclaw/skills/interview plugins/codexclaw/skills/loop/references/durable-goalplan.md` is empty, and the quoted lines (`SKILL.md:28`, `:46-53`, `:170-172`, `:176-178`, `mind-dispatch.md:47-49`, `durable-goalplan.md:40`) read as planned. No amendment; the architect's D17-D21 stand, so no re-consultation. + +Carried forward from the wp2 D summary: a CI or local failure in `subagent-config/test/spawn-attach-hook.test.ts:920` (a timing assertion) is the known flake, not a wp3 regression; diagnose it from the log before any rerun. The hypothesis that died in wp2: that dispatch prose could be written from the plan without re-reading the implemented rule. For wp3 that means C reads the final skill text, not only this doc. diff --git a/plugins/codexclaw/skills/interview/SKILL.md b/plugins/codexclaw/skills/interview/SKILL.md index f7970803..f0051876 100644 --- a/plugins/codexclaw/skills/interview/SKILL.md +++ b/plugins/codexclaw/skills/interview/SKILL.md @@ -25,7 +25,8 @@ SessionStart binding. No-FSM requests remain advisory without a transition. - Ask across four dimensions: Goal, Constraint, Success criteria, Ontology. - Re-scan contradictions after every user answer. - Do not advance to Plan while a high contradiction or pending question remains. -- Record medium/low unresolved items as OPEN ASSUMPTIONS before leaving Interview. +- Record medium/low unresolved items as OPEN ASSUMPTIONS before leaving Interview, + with the provenance fields of INTERVIEW-ASSUME-01. - When Interview reveals work that will span 2+ PABCD cycles, flag the unit as multi-cycle so that the first work-phase enters as a docs-only roadmap cycle (LOOP-DOCS-FIRST-01, `cxc-loop`). Interview settles unit residence @@ -43,6 +44,38 @@ an assumption. When evidence cannot settle a cheap, bounded comparison, offer a parallel spike and evidence-based selection. Do not invent irrelevant feature or technology choices that the project already settles. +## Assumption provenance (INTERVIEW-ASSUME-01) + +An assumption the assistant inferred is not a requirement the user agreed to, and +the handoff to Plan keeps the two apart. Write each assumption in the plan file as +one line: + + - A3 [proposed] Exports stay CSV only — source: src/export.ts:41; confidence: medium; if wrong: the XLSX writer and its tests join the scope + +- `source` is a repository `path:line`, or for something the user said, the + `eventId` of its `answer_recorded` event in the Q/A ledger + (`::answer_recorded`). A bare `questionId` is not enough; + it can repeat across turns. +- `confidence` is `low`, `medium` or `high`. `if wrong` names what changes in + scope, design or verification. +- Status is `proposed` (inferred, not yet asked), `open` (asked or deliberately + deferred, still unresolved), `user_confirmed` or `user_rejected`. The last two + require the answer's `eventId`; without one an entry stays `proposed` or + `open`, whatever the conversation seemed to imply. Under an active goal, where + Interview is suppressed, a decided goalplan decision id is the answer reference. + A reply typed in chat has no `eventId`: confirm it through the next + `request_user_input` round, or keep the entry `open` and quote the reply. +- Only `proposed` and `open` entries go under `## OPEN ASSUMPTIONS`. If a tracker + holds assumptions, the same rule applies there, because freeze carries every + recorded tracker assumption into the manifest as open; do not hand-edit session + state to create one. Move a `user_confirmed` entry into the plan's requirements with + its reference. Move a `user_rejected` entry under `## ASSUMPTION DECISIONS` with + its reference, so the decision stays traceable without being carried as open. +- Where a tracker assumption exists, its `text` repeats the same line without its + leading `- ` (freeze prepends it), so the frozen manifest keeps the provenance. +- This rule shapes existing plan text and tracker entries. It adds no field or + command, and older plans and trackers read as before. + ## Question quality (INTERVIEW-Q-01) - Target the weakest dimension first and name why it is the current bottleneck. @@ -51,6 +84,8 @@ technology choices that the project already settles. answer changes the other (INTERVIEW-INDEPENDENT-01). Independence governs, not a count. Note the transport limit: `request_user_input` accepts at most three questions per call, so a larger independent batch has to be split across calls. +- High-impact `proposed` assumptions (INTERVIEW-ASSUME-01) are candidates for the + next relevant question round. Low-impact ones may stay `proposed`; closeout lists them. - Prefer repo-grounded confirmation ("the code does X — is that intended?") over re-asking what the codebase already answers. - Treat every answer as a claim to pressure-test: vague or hedged answers do not raise a @@ -169,13 +204,16 @@ work-phase (loop-engineering §11.4). `PostToolUse` hook capture the answer, then `cxc scan record --derive --map =`. - Treat readiness as a coverage claim on top of that: each dimension has concrete knowns, no unresolved unknown changes scope, and every contradiction has exited into an answer or a - recorded assumption. Summarize the remaining OPEN ASSUMPTIONS before claiming I -> P readiness. + recorded assumption. Before claiming I -> P readiness, summarize in two groups: confirmed + requirements with their answer references, then the remaining `proposed` and `open` + assumptions with their `if wrong` consequences (INTERVIEW-ASSUME-01). ## Closeout fork (INTERVIEW-FORK-01) In non-goal HITL Interview only (under an active goal the Interview is suppressed and `request_user_input` is hard-denied — see Goal firewall), after a scan round do not drift forward -silently. Present a numbered choice and let the user pick: `1. Proceed to Plan` · +silently. Show the two-group summary from INTERVIEW-SCAN-01, then present a numbered choice and +let the user pick: `1. Proceed to Plan` · `2. Keep interviewing` · `3. Record assumptions and pause`. Do not offer a question BUDGET ("ask 2-3 more"): no tracker field persists it, so the number is unenforceable across turns, and INTERVIEW-INDEPENDENT-01 governs batching by independence rather than count. diff --git a/plugins/codexclaw/skills/interview/references/mind-dispatch.md b/plugins/codexclaw/skills/interview/references/mind-dispatch.md index fe1fd3bf..9b801f41 100644 --- a/plugins/codexclaw/skills/interview/references/mind-dispatch.md +++ b/plugins/codexclaw/skills/interview/references/mind-dispatch.md @@ -45,5 +45,6 @@ or inherit the parent, depending on the actual native/hook path. Use returned handles with the live wait/follow-up/retirement tools. Retain actual contradiction results and their evidence; malformed or missing results are not a completed independent scan. Main triages high contradictions into questions and -low/medium into OPEN ASSUMPTIONS, and records only actual authorized scan/tracker -work. The existing answer-provenance/readiness and completion gates remain intact. +low/medium into OPEN ASSUMPTIONS, which start as `proposed` under +INTERVIEW-ASSUME-01, and records only actual authorized scan/tracker work. The +existing answer-provenance/readiness and completion gates remain intact. diff --git a/plugins/codexclaw/skills/loop/references/durable-goalplan.md b/plugins/codexclaw/skills/loop/references/durable-goalplan.md index de80b491..44bbe670 100644 --- a/plugins/codexclaw/skills/loop/references/durable-goalplan.md +++ b/plugins/codexclaw/skills/loop/references/durable-goalplan.md @@ -37,7 +37,10 @@ Interview OPEN ASSUMPTIONS, steering decisions, and quality gates. ### Contract - Represent goals, work phases, success criteria, checkpoints, and evidence. -- Carry Interview OPEN ASSUMPTIONS into Plan/Audit instead of dropping them. +- Carry Interview OPEN ASSUMPTIONS into Plan/Audit instead of dropping them, with + their source, confidence, consequence if wrong and status. Only `proposed` and + `open` entries count as open; confirmed and rejected entries keep their answer + reference (INTERVIEW-ASSUME-01 in cxc-interview). - Record steering decisions with rationale and evidence. - Reject steering that weakens completion criteria or verification. - Require a quality gate before final completion. From 23b926700585ca947f39b5888976e50f0e1197b6 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Wed, 30 Sep 2026 02:52:55 +0900 Subject: [PATCH 83/90] docs(interview): clarify INTERVIEW-ASSUME-01 after fresh-reader check --- .../020_wp3_interview_assumptions.md | 6 ++- plugins/codexclaw/skills/interview/SKILL.md | 42 +++++++++++-------- .../loop/references/durable-goalplan.md | 8 ++-- 3 files changed, 34 insertions(+), 22 deletions(-) diff --git a/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md b/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md index 20106f8f..5adb74cd 100644 --- a/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md +++ b/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md @@ -116,7 +116,7 @@ SoT sync (SOT-SYNC-01): the Interview skill is the canonical owner of these rule | Check | Where it is met | |---|---| | 1. the rule requires an answer reference before an inference is presented as user-confirmed (rule-level; see enforcement naming) | 1(b) status bullet: confirmed/rejected require the answer `eventId` | -| 2. rejected inference kept as decision trace, not carried as open | 1(b) last-but-one bullet: `## ASSUMPTION DECISIONS`, never in tracker assumptions | +| 2. rejected inference kept as decision trace, not carried as open | 1(b) three-sections bullet and tracker bullet: rejected entries go under `## ASSUMPTION DECISIONS`; only proposed/open entries belong in the tracker | | 3. closeout distinguishes confirmed requirements from open inferred assumptions | 1(d), 1(e) | | 4. existing trackers and freeze manifests remain readable | no code change; 1(b) last bullet | @@ -140,3 +140,7 @@ Tier E7 (agent-followed guidance). Executing surface: the main session writing t Continuity (LOOP-CONTINUITY-01), quoting the wp2 D summary in 010: "#276 and #277 are fixed and merged ... Next: wp3 builds 020." This P keeps that direction. Re-checked on `codex/issue-train-0930-wp3` from `origin/dev` `069a7d0e`: `git diff --stat 659de59b..HEAD -- plugins/codexclaw/skills/interview plugins/codexclaw/skills/loop/references/durable-goalplan.md` is empty, and the quoted lines (`SKILL.md:28`, `:46-53`, `:170-172`, `:176-178`, `mind-dispatch.md:47-49`, `durable-goalplan.md:40`) read as planned. No amendment; the architect's D17-D21 stand, so no re-consultation. Carried forward from the wp2 D summary: a CI or local failure in `subagent-config/test/spawn-attach-hook.test.ts:920` (a timing assertion) is the known flake, not a wp3 regression; diagnose it from the log before any rerun. The hypothesis that died in wp2: that dispatch prose could be written from the plan without re-reading the implemented rule. For wp3 that means C reads the final skill text, not only this doc. + +## wp3 C record (2026-09-30) + +C round 1 on `4254bc38`: fresh implementation reviewer `01a0ee49-6e1e` GO-WITH-FIXES (blockers=0), initiative verifier `01a0ee49-6f1d` GO-WITH-FIXES (4: reader check, semantic review of the final text, check output with the `rg` result, goalplan/delivery). The reviewer's C-READER-01 pass read only the new section and the SCAN/FORK lines and stumbled at: the unexplained id and bracket in the example line; when the goal-mode sentence applies; "tracker" and "freeze" undefined; no heading for confirmed requirements; "high-impact" not tied to a field; the goalplan sentence mixing open and resolved entries; plus an `eventId` caveat (`no-turn` repeats) and no way out for a resolved tracker entry. Changes: the section now explains the line format, names three plan sections (`## OPEN ASSUMPTIONS`, `## CONFIRMED REQUIREMENTS`, `## ASSUMPTION DECISIONS`), defines the tracker and `cxc freeze`, ties high-impact to `if wrong`, adds the `no-turn:` caveat, makes the plan line authoritative when a tracker entry cannot be moved, and says a goal-mode decision's answer quotes the user's reply; `durable-goalplan.md:40` separates open entries from the resolved sections. The acceptance table's pointer for check 2 was corrected. The goal-mode reference remains agent-recorded (named in enforcement naming as part of the same bypass). diff --git a/plugins/codexclaw/skills/interview/SKILL.md b/plugins/codexclaw/skills/interview/SKILL.md index f0051876..7d5e1e55 100644 --- a/plugins/codexclaw/skills/interview/SKILL.md +++ b/plugins/codexclaw/skills/interview/SKILL.md @@ -47,32 +47,40 @@ technology choices that the project already settles. ## Assumption provenance (INTERVIEW-ASSUME-01) An assumption the assistant inferred is not a requirement the user agreed to, and -the handoff to Plan keeps the two apart. Write each assumption in the plan file as -one line: +the handoff to Plan keeps the two apart. In the plan file, write each assumption +as one line: an id (`A1`, `A2`, ...), its status in brackets, the assumption, +then its source, confidence and consequence if wrong. - A3 [proposed] Exports stay CSV only — source: src/export.ts:41; confidence: medium; if wrong: the XLSX writer and its tests join the scope - `source` is a repository `path:line`, or for something the user said, the `eventId` of its `answer_recorded` event in the Q/A ledger - (`::answer_recorded`). A bare `questionId` is not enough; - it can repeat across turns. + (`::answer_recorded`). A bare `questionId` is not enough + because it can repeat across turns; an `eventId` starting with `no-turn:` has + the same weakness, so re-ask rather than rely on it. - `confidence` is `low`, `medium` or `high`. `if wrong` names what changes in - scope, design or verification. + scope, design or verification; an assumption is high-impact when that + consequence changes scope or verification. - Status is `proposed` (inferred, not yet asked), `open` (asked or deliberately deferred, still unresolved), `user_confirmed` or `user_rejected`. The last two require the answer's `eventId`; without one an entry stays `proposed` or - `open`, whatever the conversation seemed to imply. Under an active goal, where - Interview is suppressed, a decided goalplan decision id is the answer reference. - A reply typed in chat has no `eventId`: confirm it through the next - `request_user_input` round, or keep the entry `open` and quote the reply. -- Only `proposed` and `open` entries go under `## OPEN ASSUMPTIONS`. If a tracker - holds assumptions, the same rule applies there, because freeze carries every - recorded tracker assumption into the manifest as open; do not hand-edit session - state to create one. Move a `user_confirmed` entry into the plan's requirements with - its reference. Move a `user_rejected` entry under `## ASSUMPTION DECISIONS` with - its reference, so the decision stays traceable without being carried as open. -- Where a tracker assumption exists, its `text` repeats the same line without its - leading `- ` (freeze prepends it), so the frozen manifest keeps the provenance. + `open`, whatever the conversation seemed to imply. A reply typed in chat has + no `eventId`: confirm it through the next `request_user_input` round, or keep + the entry `open` and quote the reply. +- The plan keeps three sections. `## OPEN ASSUMPTIONS` holds only `proposed` and + `open` entries. `## CONFIRMED REQUIREMENTS` holds `user_confirmed` entries + with their reference. `## ASSUMPTION DECISIONS` holds `user_rejected` entries + with their reference, so the decision stays traceable without being carried as + open. +- The Interview tracker (session state read by the readiness gate) may also hold + assumptions, and `cxc freeze` copies every recorded one into the frozen + manifest as open, adding the leading `- `. Where a tracker entry exists, its + `text` repeats the plan line without that `- `, and only `proposed` or `open` + entries belong there. Do not hand-edit session state; if a tracker entry + cannot be moved after it is resolved, the plan line's status is authoritative. +- A plan written later under an active goal, where Interview is suppressed, may + use a decided goalplan decision id (`cxc loop decide`) as the reference, with the + decision's answer quoting the user's reply. - This rule shapes existing plan text and tracker entries. It adds no field or command, and older plans and trackers read as before. diff --git a/plugins/codexclaw/skills/loop/references/durable-goalplan.md b/plugins/codexclaw/skills/loop/references/durable-goalplan.md index 44bbe670..822d998f 100644 --- a/plugins/codexclaw/skills/loop/references/durable-goalplan.md +++ b/plugins/codexclaw/skills/loop/references/durable-goalplan.md @@ -37,10 +37,10 @@ Interview OPEN ASSUMPTIONS, steering decisions, and quality gates. ### Contract - Represent goals, work phases, success criteria, checkpoints, and evidence. -- Carry Interview OPEN ASSUMPTIONS into Plan/Audit instead of dropping them, with - their source, confidence, consequence if wrong and status. Only `proposed` and - `open` entries count as open; confirmed and rejected entries keep their answer - reference (INTERVIEW-ASSUME-01 in cxc-interview). +- Carry Interview OPEN ASSUMPTIONS (`proposed` and `open` entries, with source, + confidence, consequence if wrong and status) into Plan/Audit instead of dropping + them. Confirmed requirements and rejected assumptions travel in their own plan + sections with their answer reference (INTERVIEW-ASSUME-01 in cxc-interview). - Record steering decisions with rationale and evidence. - Reject steering that weakens completion criteria or verification. - Require a quality gate before final completion. From 6e7bd3eb41cb39a9e07696f19ef59d1c009c815c Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Wed, 30 Sep 2026 02:56:42 +0900 Subject: [PATCH 84/90] docs(interview): answer reference wording, high-impact includes design --- .../020_wp3_interview_assumptions.md | 2 ++ plugins/codexclaw/skills/interview/SKILL.md | 11 ++++++----- 2 files changed, 8 insertions(+), 5 deletions(-) diff --git a/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md b/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md index 5adb74cd..0ba60b20 100644 --- a/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md +++ b/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md @@ -144,3 +144,5 @@ Carried forward from the wp2 D summary: a CI or local failure in `subagent-confi ## wp3 C record (2026-09-30) C round 1 on `4254bc38`: fresh implementation reviewer `01a0ee49-6e1e` GO-WITH-FIXES (blockers=0), initiative verifier `01a0ee49-6f1d` GO-WITH-FIXES (4: reader check, semantic review of the final text, check output with the `rg` result, goalplan/delivery). The reviewer's C-READER-01 pass read only the new section and the SCAN/FORK lines and stumbled at: the unexplained id and bracket in the example line; when the goal-mode sentence applies; "tracker" and "freeze" undefined; no heading for confirmed requirements; "high-impact" not tied to a field; the goalplan sentence mixing open and resolved entries; plus an `eventId` caveat (`no-turn` repeats) and no way out for a resolved tracker entry. Changes: the section now explains the line format, names three plan sections (`## OPEN ASSUMPTIONS`, `## CONFIRMED REQUIREMENTS`, `## ASSUMPTION DECISIONS`), defines the tracker and `cxc freeze`, ties high-impact to `if wrong`, adds the `no-turn:` caveat, makes the plan line authoritative when a tracker entry cannot be moved, and says a goal-mode decision's answer quotes the user's reply; `durable-goalplan.md:40` separates open entries from the resolved sections. The acceptance table's pointer for check 2 was corrected. The goal-mode reference remains agent-recorded (named in enforcement naming as part of the same bypass). + +C round 2 on `23b92670`: reviewer GO-WITH-FIXES (blockers=0) after a second fresh-reader pass, initiative verifier PASS; C gate OK on `23b92670` (3730 tests, 0 failures). Three wording fixes both flagged were applied in the next commit (answer reference covers the goal-mode decision id in the status bullet; high-impact includes design; "after Interview" and the `source` field for the goal-mode reference). The final skill text supersedes plan item 1(b) above. diff --git a/plugins/codexclaw/skills/interview/SKILL.md b/plugins/codexclaw/skills/interview/SKILL.md index 7d5e1e55..33b30dd1 100644 --- a/plugins/codexclaw/skills/interview/SKILL.md +++ b/plugins/codexclaw/skills/interview/SKILL.md @@ -60,10 +60,11 @@ then its source, confidence and consequence if wrong. the same weakness, so re-ask rather than rely on it. - `confidence` is `low`, `medium` or `high`. `if wrong` names what changes in scope, design or verification; an assumption is high-impact when that - consequence changes scope or verification. + consequence changes any of them. - Status is `proposed` (inferred, not yet asked), `open` (asked or deliberately deferred, still unresolved), `user_confirmed` or `user_rejected`. The last two - require the answer's `eventId`; without one an entry stays `proposed` or + require an answer reference: the answer's `eventId`, or under an active goal a + decided goalplan decision id (below). Without one an entry stays `proposed` or `open`, whatever the conversation seemed to imply. A reply typed in chat has no `eventId`: confirm it through the next `request_user_input` round, or keep the entry `open` and quote the reply. @@ -78,9 +79,9 @@ then its source, confidence and consequence if wrong. `text` repeats the plan line without that `- `, and only `proposed` or `open` entries belong there. Do not hand-edit session state; if a tracker entry cannot be moved after it is resolved, the plan line's status is authoritative. -- A plan written later under an active goal, where Interview is suppressed, may - use a decided goalplan decision id (`cxc loop decide`) as the reference, with the - decision's answer quoting the user's reply. +- A plan written after Interview, under an active goal where Interview is + suppressed, may put a decided goalplan decision id (`cxc loop decide`) in the + `source` field, with the decision's answer quoting the user's reply. - This rule shapes existing plan text and tracker entries. It adds no field or command, and older plans and trackers read as before. From f67c55bf3571b762672c803a075e9911cd515675 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Wed, 30 Sep 2026 03:10:14 +0900 Subject: [PATCH 85/90] docs(plan): wp3 D summary; wp5 P revalidation --- .../260930_issue_train/020_wp3_interview_assumptions.md | 6 ++++++ .../_plan/260930_issue_train/030_wp5_decision_options.md | 8 ++++++++ 2 files changed, 14 insertions(+) diff --git a/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md b/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md index 0ba60b20..b0883723 100644 --- a/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md +++ b/devlog/_plan/260930_issue_train/020_wp3_interview_assumptions.md @@ -146,3 +146,9 @@ Carried forward from the wp2 D summary: a CI or local failure in `subagent-confi C round 1 on `4254bc38`: fresh implementation reviewer `01a0ee49-6e1e` GO-WITH-FIXES (blockers=0), initiative verifier `01a0ee49-6f1d` GO-WITH-FIXES (4: reader check, semantic review of the final text, check output with the `rg` result, goalplan/delivery). The reviewer's C-READER-01 pass read only the new section and the SCAN/FORK lines and stumbled at: the unexplained id and bracket in the example line; when the goal-mode sentence applies; "tracker" and "freeze" undefined; no heading for confirmed requirements; "high-impact" not tied to a field; the goalplan sentence mixing open and resolved entries; plus an `eventId` caveat (`no-turn` repeats) and no way out for a resolved tracker entry. Changes: the section now explains the line format, names three plan sections (`## OPEN ASSUMPTIONS`, `## CONFIRMED REQUIREMENTS`, `## ASSUMPTION DECISIONS`), defines the tracker and `cxc freeze`, ties high-impact to `if wrong`, adds the `no-turn:` caveat, makes the plan line authoritative when a tracker entry cannot be moved, and says a goal-mode decision's answer quotes the user's reply; `durable-goalplan.md:40` separates open entries from the resolved sections. The acceptance table's pointer for check 2 was corrected. The goal-mode reference remains agent-recorded (named in enforcement naming as part of the same bypass). C round 2 on `23b92670`: reviewer GO-WITH-FIXES (blockers=0) after a second fresh-reader pass, initiative verifier PASS; C gate OK on `23b92670` (3730 tests, 0 failures). Three wording fixes both flagged were applied in the next commit (answer reference covers the goal-mode decision id in the status bullet; high-impact includes design; "after Interview" and the `source` field for the goal-mode reference). The final skill text supersedes plan item 1(b) above. + +## wp3 D summary (2026-09-30) + +Conclusion: #275 is addressed at the rule level and merged into `dev` through PR #279 (head `6e7bd3eb`, 14/14 checks, merge `58a8a174`). Evidence: INTERVIEW-ASSUME-01 referenced in the three files, two C rounds with fresh-reader passes whose stumbles changed the text, and the C gate on the final head (3730 tests, 0 failures). Next: wp5 builds 030. + +What did not go well: the planned prose (1(b)) was written for the plan reviewers, not for a first-time reader; the fresh-reader pass found seven stumbles, and the final text differs materially from the plan. The rule stays agent-followed: nothing checks an `eventId` or decision id, and the goal-mode reference is recorded by the agent. The hypothesis that died: that guidance audited at A reads well to its real audience. Evidence that the direction is wrong: agents keep presenting inferred assumptions as confirmed despite the rule, which would argue for the structured schema the issue lists as future work. diff --git a/devlog/_plan/260930_issue_train/030_wp5_decision_options.md b/devlog/_plan/260930_issue_train/030_wp5_decision_options.md index 952ecf68..a52a7913 100644 --- a/devlog/_plan/260930_issue_train/030_wp5_decision_options.md +++ b/devlog/_plan/260930_issue_train/030_wp5_decision_options.md @@ -160,3 +160,11 @@ Activation scenarios (C-ACTIVATION-GROUNDING-01): (2d)'s blank and repeated `--o ## Enforcement naming (PLAN-BYPASS-NAMED-01) Tier E2 (CLI and reviver validation). Executing surface: `cxc loop ask` and every goalplan read. Known bypass: a library caller writing through `writeGoalplan` directly skips `ask`'s checks, but the next read still fails closed on invalid options; nothing proves the question was actually sent with those options. Residual risk: an invalid hand edit makes the whole plan unreadable until repaired, as for every existing field. Wording: validation, not enforcement of what the host displayed. Final enforcement layer: none. + +## wp5 P revalidation (2026-09-30) + +Continuity (LOOP-CONTINUITY-01), quoting the wp3 D summary in 020: "#275 is addressed at the rule level and merged ... Next: wp5 builds 030." This P keeps that direction, and carries forward wp2's known timing flake (`spawn-attach-hook.test.ts:920`) and wp3's lesson that prose written into docs must be re-read in its final form at C. + +Re-checked on `codex/issue-train-0930-wp5` from `origin/dev` `58a8a174`: `git diff --stat 659de59b..HEAD -- plugins/codexclaw/components/pabcd-state plugins/codexclaw/skills/dev/references/async-questions.md` is empty, so every `goalplan.ts` and `goalplan-cli.ts` anchor above holds. As planned, wp3's three-line edit moved the `durable-goalplan.md` anchors: the `decisions[]` schema line is now `:63` (was `:60`) and the `ask` synopsis is `:101` (was `:98`). `async-questions.md:56` is unchanged. `renderGoalplanHelp` (`goalplan-cli.ts:702`) prints the verb usage strings, so it picks up the new `ask` usage without a separate edit. No design decision changes, so no architect recheck. + +Amendment (architect's optional D26 note): add one Notes line to `renderGoalplanHelp` after the `decide` note (`goalplan-cli.ts:724`): `" Repeat --option once per offered option; the recommendation must be one of them, and the answer stays free text."` No test pins the help Notes text (checked with `rg 'Record the user' plugins/codexclaw/components/pabcd-state/test`). From 355afde3f5479b66a9b2ea78355e17dd98d5abe8 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Wed, 30 Sep 2026 03:13:19 +0900 Subject: [PATCH 86/90] feat(goalplan): decision options; recommendation must be one of them (#262 follow-up) --- .../030_wp5_decision_options.md | 6 +- .../pabcd-state/dist/goalplan-cli.js | 21 +++- .../components/pabcd-state/dist/goalplan.js | 37 +++++- .../pabcd-state/src/goalplan-cli.ts | 23 +++- .../components/pabcd-state/src/goalplan.ts | 37 +++++- .../test/goalplan-public-surface.test.ts | 108 +++++++++++++++++- .../skills/dev/references/async-questions.md | 2 +- .../loop/references/durable-goalplan.md | 4 +- 8 files changed, 219 insertions(+), 19 deletions(-) diff --git a/devlog/_plan/260930_issue_train/030_wp5_decision_options.md b/devlog/_plan/260930_issue_train/030_wp5_decision_options.md index a52a7913..06ca5864 100644 --- a/devlog/_plan/260930_issue_train/030_wp5_decision_options.md +++ b/devlog/_plan/260930_issue_train/030_wp5_decision_options.md @@ -124,6 +124,8 @@ and in both `decisions.push` calls (`:539-540`, `:543-545`) append `...(options + } ``` +(h) Help Notes line after the `decide` note (`:724`): "Repeat --option once per offered option; the recommendation must be one of them, and the answer stays free text." (A round nit: moved into the file map from the revalidation.) + ### 3. REGENERATE `pabcd-state/dist/goalplan.js`, `pabcd-state/dist/goalplan-cli.js` (`npm run build`; tracked, checked by `dist-freshness.test.mjs`). ### 4. MODIFY `plugins/codexclaw/components/pabcd-state/test/goalplan-public-surface.test.ts` @@ -134,7 +136,7 @@ Beside the existing decision tests (`:684-764`), using the same temp-dir CLI hel |---|---|---| | `ask records options and ready/show expose them` | `ask --option A --option B --recommendation A` | goalplan decision has `options:["A","B"]`; `ready --json` open decision has `options`; `show` prints `options: A \| B (recommended: A)` | | `ask rejects a recommendation outside the options without a write` | `--option A --option B --recommendation C` | exit 1, reason `must be one of the options`, plan bytes unchanged | -| `ask rejects blank and repeated options at parse time` | `--option " "`; `--option A --option A` | exit 1 with the parser reasons, no write | +| `ask rejects blank and repeated options at parse time` | `--option " "`; `--option A --option A`; `--option` on `decide` | `parseGoalplanCliArgs` returns the parser error (the `cli()` helper asserts a successful parse), plan bytes unchanged | | `ask without --option stores no options key` | plain ask | decision JSON has no `options` property | | `decide keeps options and accepts a free-form answer` | ask with options, `decide --answer "something else"` | exit 0, decision decided, `options` unchanged | | `reviver fails closed on malformed options` | hand-written plans with `options: {}`, `options: []`, `[" "]`, `["A","A "]`, `[1]`, and a recommendation outside valid options | each read fails naming the field: `field 'decisions' did not satisfy the schema` (`goalplan.ts:719` via `firstInvalidField`), before `validateGoalplan` runs | @@ -143,7 +145,7 @@ Beside the existing decision tests (`:684-764`), using the same temp-dir CLI hel ### 5. MODIFY docs - `plugins/codexclaw/skills/loop/references/durable-goalplan.md:60`: `{ id, question, recommendation?, options?, status: open|decided, answer?, askedAt, decidedAt? }` and one sentence: "When `options` is present it is a non-empty list of distinct entries and `recommendation` must be one of them; the answer stays free text because the host always offers a free-form reply." -- `durable-goalplan.md:98`: add `[--option ]...` to the `ask` synopsis. +- `durable-goalplan.md:101` (was `:98`): add `[--option ]...` to the `ask` synopsis. - `plugins/codexclaw/skills/dev/references/async-questions.md:56`: add `[--option ]...` to the `ask` synopsis and "(the offered options, recommended first)". ## Verification (PLAN-VERIFIER-REAL-01) diff --git a/plugins/codexclaw/components/pabcd-state/dist/goalplan-cli.js b/plugins/codexclaw/components/pabcd-state/dist/goalplan-cli.js index 17aefe6f..ef7e4025 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/goalplan-cli.js +++ b/plugins/codexclaw/components/pabcd-state/dist/goalplan-cli.js @@ -119,6 +119,8 @@ import { applySteeringBatch } from "./steering.js"; + + @@ -160,7 +162,7 @@ const VERB_RULES = { "add-task": { allowed: new Set(["--session", "--work-phase", "--id", "--title", "--depends-on", "--cwd"]), repeatable: new Set(["--depends-on"]), usage: "add-task --session --work-phase --id --title [--depends-on ]... [--cwd ]" }, "complete-task": { allowed: new Set(["--session", "--work-phase", "--id", "--outcome", "--cwd"]), repeatable: new Set(), usage: "complete-task --session --work-phase --id --outcome [--cwd ]" }, "meet-criterion": { allowed: new Set(["--session", "--id", "--evidence", "--cwd"]), repeatable: new Set(), usage: "meet-criterion --session --id --evidence [--cwd ]" }, - ask: { allowed: new Set(["--session", "--id", "--question", "--recommendation", "--work-phase", "--cwd"]), repeatable: new Set(["--work-phase"]), usage: "ask --session --id --question [--recommendation ] [--work-phase ]... [--cwd ]" }, + ask: { allowed: new Set(["--session", "--id", "--question", "--recommendation", "--option", "--work-phase", "--cwd"]), repeatable: new Set(["--option", "--work-phase"]), usage: "ask --session --id --question [--recommendation ] [--option ]... [--work-phase ]... [--cwd ]" }, decide: { allowed: new Set(["--session", "--id", "--answer", "--cwd"]), repeatable: new Set(), usage: "decide --session --id --answer [--cwd ]" }, help: { allowed: new Set(), repeatable: new Set(), usage: "--help" }, }; @@ -237,6 +239,14 @@ export function parseGoalplanCliArgs(argv , cwd ) } case "--question": out.question = value; break; case "--recommendation": out.recommendation = value; break; + case "--option": { + const option = value.trim(); + if (!option) return reject("--option requires one non-empty value"); + const options = out.options ?? (out.options = []); + if (options.includes(option)) return reject(`--option must not repeat '${option}'`); + options.push(option); + break; + } case "--answer": out.answer = value; break; case "--outcome": out.outcome = value; break; case "--schema-version": { @@ -463,7 +473,9 @@ function runReady(args , plan ) { const phases = readyWorkPhases(plan); const tasks = readyTasks(plan); const openDecisions = (plan.decisions ?? []).filter((decision) => decision.status === "open") - .map(({ id, question, recommendation, askedAt }) => ({ id, question, ...(recommendation === undefined ? {} : { recommendation }), askedAt })); + .map(({ id, question, recommendation, options, askedAt }) => ({ id, question, + ...(recommendation === undefined ? {} : { recommendation }), + ...(options === undefined ? {} : { options }), askedAt })); const awaitingDecisions = plan.workPhases .filter((wp) => wp.status === "pending" || wp.status === "in_progress") .map((wp) => ({ workPhaseId: wp.id, decisionIds: openDecisionIdsForPhase(plan, wp).filter((id) => @@ -529,6 +541,7 @@ function runDecision(args ) { const result = args.verb === "ask" ? askGoalplanDecision(plan, { id, question: args.question , recommendation: args.recommendation, + ...(args.options === undefined ? {} : { options: args.options }), workPhaseIds: args.workPhaseIds ?? [], askedAt: new Date().toISOString(), }) : decideGoalplanDecision(plan, id, args.answer , new Date().toISOString()); @@ -688,6 +701,9 @@ function renderPlanLines(plan , lock ) for (const decision of plan.decisions ?? []) { if (decision.status !== "open") continue; lines.push(` - ${decision.id} [open] ${decision.question}`); + if (decision.options !== undefined) { + lines.push(` options: ${decision.options.join(" | ")}${decision.recommendation === undefined ? "" : ` (recommended: ${decision.recommendation})`}`); + } const waiting = plan.workPhases.filter((wp) => wp.awaitsDecision?.includes(decision.id)); lines.push(` waiting: ${waiting.map((wp) => wp.id).join(", ") || "none"}`); } @@ -722,6 +738,7 @@ export function renderGoalplanHelp() { " meet-criterion requires non-empty captured evidence for the same reason.", " Send the question through the host first, then record it with ask; ask never sends a message.", " Record the user's reply with decide. It changes only the decision record.", + " Repeat --option once per offered option; the recommendation must be one of them, and the answer stays free text.", "", "steer --batch-json expects an object with:", ' { "idempotencyKey": "", "rationale": "", "evidence": "",', diff --git a/plugins/codexclaw/components/pabcd-state/dist/goalplan.js b/plugins/codexclaw/components/pabcd-state/dist/goalplan.js index c03c80a4..72441c05 100644 --- a/plugins/codexclaw/components/pabcd-state/dist/goalplan.js +++ b/plugins/codexclaw/components/pabcd-state/dist/goalplan.js @@ -160,6 +160,8 @@ export const DEFAULT_NEW_SCHEMA_VERSION = 1; + + @@ -523,6 +525,16 @@ function validIsoTime(value ) { return Number.isFinite(date.valueOf()) && date.toISOString() === value; } +/** Absent stays absent; a present list must be non-empty, non-blank and distinct after trim. */ +function reviveDecisionOptions(value ) { + if (value === undefined) return undefined; + if (!Array.isArray(value) || value.length === 0) return "invalid"; + if (value.some((option) => typeof option !== "string" || !option.trim())) return "invalid"; + const trimmed = (value ).map((option) => option.trim()); + if (new Set(trimmed).size !== trimmed.length) return "invalid"; + return [...(value )]; +} + function reviveDecisions(value ) { if (value === undefined) return undefined; if (!Array.isArray(value)) return "invalid"; @@ -534,15 +546,21 @@ function reviveDecisions(value ) || typeof d.question !== "string" || !d.question.trim() || !validIsoTime(d.askedAt) || (d.recommendation !== undefined && (typeof d.recommendation !== "string" || !d.recommendation.trim()))) return "invalid"; + const options = reviveDecisionOptions(d.options); + if (options === "invalid") return "invalid"; + if (options !== undefined && d.recommendation !== undefined + && !options.some((option) => option.trim() === (d.recommendation ).trim())) return "invalid"; if (d.status === "open") { if (d.answer !== undefined || d.decidedAt !== undefined) return "invalid"; decisions.push({ id: d.id, question: d.question, status: "open", askedAt: d.askedAt, - ...(d.recommendation === undefined ? {} : { recommendation: d.recommendation }) }); + ...(d.recommendation === undefined ? {} : { recommendation: d.recommendation }), + ...(options === undefined ? {} : { options }) }); } else if (d.status === "decided") { if (typeof d.answer !== "string" || !d.answer.trim() || !validIsoTime(d.decidedAt)) return "invalid"; decisions.push({ id: d.id, question: d.question, status: "decided", answer: d.answer, askedAt: d.askedAt, decidedAt: d.decidedAt, - ...(d.recommendation === undefined ? {} : { recommendation: d.recommendation }) }); + ...(d.recommendation === undefined ? {} : { recommendation: d.recommendation }), + ...(options === undefined ? {} : { options }) }); } else return "invalid"; } return decisions; @@ -1224,7 +1242,7 @@ const LIFECYCLE_ID_RE = /^[a-z0-9][a-z0-9-]{0,39}$/; export function askGoalplanDecision( plan , - input , + input , ) { const id = input.id.trim(); const question = input.question.trim(); @@ -1233,6 +1251,16 @@ export function askGoalplanDecision( if (!LIFECYCLE_ID_RE.test(id)) return { kind: "rejected", reason: "decision id must be a short lowercase id, e.g. dec-1" }; if (!question) return { kind: "rejected", reason: "decision question must not be empty" }; if (input.recommendation !== undefined && !recommendation) return { kind: "rejected", reason: "decision recommendation must not be empty" }; + const options = input.options?.map((option) => option.trim()); + if (options !== undefined) { + if (options.length === 0) return { kind: "rejected", reason: "decision options must not be empty" }; + if (options.some((option) => !option)) return { kind: "rejected", reason: "decision options must be non-empty text" }; + const repeated = options.find((option, index) => options.indexOf(option) !== index); + if (repeated !== undefined) return { kind: "rejected", reason: `duplicate decision option '${repeated}'` }; + if (recommendation !== undefined && !options.includes(recommendation)) { + return { kind: "rejected", reason: "decision recommendation must be one of the options" }; + } + } if (!validIsoTime(input.askedAt)) return { kind: "rejected", reason: "decision askedAt must be an ISO timestamp" }; if (plan.decisions?.some((decision) => decision.id === id)) return { kind: "rejected", reason: `decision '${id}' is already in this plan` }; const duplicate = plan.decisions?.find((decision) => decision.status === "open" && decision.question.trim() === question); @@ -1248,7 +1276,8 @@ export function askGoalplanDecision( } } const decision = { id, question, status: "open", askedAt: input.askedAt, - ...(recommendation === undefined ? {} : { recommendation }) }; + ...(recommendation === undefined ? {} : { recommendation }), + ...(options === undefined ? {} : { options }) }; const next = { ...plan, decisions: [...(plan.decisions ?? []), decision], workPhases: plan.workPhases.map((wp) => workPhaseIds.includes(wp.id) ? { ...wp, awaitsDecision: [...(wp.awaitsDecision ?? []), id] } : wp) }; diff --git a/plugins/codexclaw/components/pabcd-state/src/goalplan-cli.ts b/plugins/codexclaw/components/pabcd-state/src/goalplan-cli.ts index e6d584dc..1780ec2e 100644 --- a/plugins/codexclaw/components/pabcd-state/src/goalplan-cli.ts +++ b/plugins/codexclaw/components/pabcd-state/src/goalplan-cli.ts @@ -106,6 +106,8 @@ export interface GoalplanCliArgs { workPhaseIds?: string[]; question?: string; recommendation?: string; + /** `ask`: options offered with the question; absent unless --option was given. */ + options?: string[]; answer?: string; /** `complete-task`: the outcome evidence a done task must carry. */ outcome?: string; @@ -141,7 +143,7 @@ type GoalplanFlag = | "--objective" | "--slug" | "--criterion" | "--cwd" | "--session" | "--batch-json" | "--surface" | "--presented" | "--id" | "--title" | "--work-phase" | "--outcome" | "--schema-version" | "--evidence" | "--json" | "--depends-on" - | "--question" | "--recommendation" | "--answer"; + | "--question" | "--recommendation" | "--option" | "--answer"; type VerbRule = { allowed: ReadonlySet; @@ -160,7 +162,7 @@ const VERB_RULES: Readonly> = { "add-task": { allowed: new Set(["--session", "--work-phase", "--id", "--title", "--depends-on", "--cwd"]), repeatable: new Set(["--depends-on"]), usage: "add-task --session --work-phase --id --title [--depends-on ]... [--cwd ]" }, "complete-task": { allowed: new Set(["--session", "--work-phase", "--id", "--outcome", "--cwd"]), repeatable: new Set(), usage: "complete-task --session --work-phase --id --outcome [--cwd ]" }, "meet-criterion": { allowed: new Set(["--session", "--id", "--evidence", "--cwd"]), repeatable: new Set(), usage: "meet-criterion --session --id --evidence [--cwd ]" }, - ask: { allowed: new Set(["--session", "--id", "--question", "--recommendation", "--work-phase", "--cwd"]), repeatable: new Set(["--work-phase"]), usage: "ask --session --id --question [--recommendation ] [--work-phase ]... [--cwd ]" }, + ask: { allowed: new Set(["--session", "--id", "--question", "--recommendation", "--option", "--work-phase", "--cwd"]), repeatable: new Set(["--option", "--work-phase"]), usage: "ask --session --id --question [--recommendation ] [--option ]... [--work-phase ]... [--cwd ]" }, decide: { allowed: new Set(["--session", "--id", "--answer", "--cwd"]), repeatable: new Set(), usage: "decide --session --id --answer [--cwd ]" }, help: { allowed: new Set(), repeatable: new Set(), usage: "--help" }, }; @@ -237,6 +239,14 @@ export function parseGoalplanCliArgs(argv: string[], cwd: string): GoalplanCliAr } case "--question": out.question = value; break; case "--recommendation": out.recommendation = value; break; + case "--option": { + const option = value.trim(); + if (!option) return reject("--option requires one non-empty value"); + const options = out.options ?? (out.options = []); + if (options.includes(option)) return reject(`--option must not repeat '${option}'`); + options.push(option); + break; + } case "--answer": out.answer = value; break; case "--outcome": out.outcome = value; break; case "--schema-version": { @@ -463,7 +473,9 @@ function runReady(args: GoalplanCliArgs, plan: Goalplan): GoalplanCliResult { const phases = readyWorkPhases(plan); const tasks = readyTasks(plan); const openDecisions = (plan.decisions ?? []).filter((decision) => decision.status === "open") - .map(({ id, question, recommendation, askedAt }) => ({ id, question, ...(recommendation === undefined ? {} : { recommendation }), askedAt })); + .map(({ id, question, recommendation, options, askedAt }) => ({ id, question, + ...(recommendation === undefined ? {} : { recommendation }), + ...(options === undefined ? {} : { options }), askedAt })); const awaitingDecisions = plan.workPhases .filter((wp) => wp.status === "pending" || wp.status === "in_progress") .map((wp) => ({ workPhaseId: wp.id, decisionIds: openDecisionIdsForPhase(plan, wp).filter((id) => @@ -529,6 +541,7 @@ function runDecision(args: GoalplanCliArgs): GoalplanCliResult { const result = args.verb === "ask" ? askGoalplanDecision(plan, { id, question: args.question!, recommendation: args.recommendation, + ...(args.options === undefined ? {} : { options: args.options }), workPhaseIds: args.workPhaseIds ?? [], askedAt: new Date().toISOString(), }) : decideGoalplanDecision(plan, id, args.answer!, new Date().toISOString()); @@ -688,6 +701,9 @@ function renderPlanLines(plan: Goalplan, lock?: GoalplanWriteLockStatus): string for (const decision of plan.decisions ?? []) { if (decision.status !== "open") continue; lines.push(` - ${decision.id} [open] ${decision.question}`); + if (decision.options !== undefined) { + lines.push(` options: ${decision.options.join(" | ")}${decision.recommendation === undefined ? "" : ` (recommended: ${decision.recommendation})`}`); + } const waiting = plan.workPhases.filter((wp) => wp.awaitsDecision?.includes(decision.id)); lines.push(` waiting: ${waiting.map((wp) => wp.id).join(", ") || "none"}`); } @@ -722,6 +738,7 @@ export function renderGoalplanHelp(): string { " meet-criterion requires non-empty captured evidence for the same reason.", " Send the question through the host first, then record it with ask; ask never sends a message.", " Record the user's reply with decide. It changes only the decision record.", + " Repeat --option once per offered option; the recommendation must be one of them, and the answer stays free text.", "", "steer --batch-json expects an object with:", ' { "idempotencyKey": "", "rationale": "", "evidence": "",', diff --git a/plugins/codexclaw/components/pabcd-state/src/goalplan.ts b/plugins/codexclaw/components/pabcd-state/src/goalplan.ts index 92bc5e20..3164d062 100644 --- a/plugins/codexclaw/components/pabcd-state/src/goalplan.ts +++ b/plugins/codexclaw/components/pabcd-state/src/goalplan.ts @@ -142,6 +142,8 @@ export interface GoalplanDecision { id: string; question: string; recommendation?: string; + /** Options offered with the question; when present, recommendation is one of them (#262). */ + options?: string[]; status: "open" | "decided"; answer?: string; askedAt: string; @@ -523,6 +525,16 @@ function validIsoTime(value: unknown): value is string { return Number.isFinite(date.valueOf()) && date.toISOString() === value; } +/** Absent stays absent; a present list must be non-empty, non-blank and distinct after trim. */ +function reviveDecisionOptions(value: unknown): string[] | undefined | "invalid" { + if (value === undefined) return undefined; + if (!Array.isArray(value) || value.length === 0) return "invalid"; + if (value.some((option) => typeof option !== "string" || !option.trim())) return "invalid"; + const trimmed = (value as string[]).map((option) => option.trim()); + if (new Set(trimmed).size !== trimmed.length) return "invalid"; + return [...(value as string[])]; +} + function reviveDecisions(value: unknown): GoalplanDecision[] | undefined | "invalid" { if (value === undefined) return undefined; if (!Array.isArray(value)) return "invalid"; @@ -534,15 +546,21 @@ function reviveDecisions(value: unknown): GoalplanDecision[] | undefined | "inva || typeof d.question !== "string" || !d.question.trim() || !validIsoTime(d.askedAt) || (d.recommendation !== undefined && (typeof d.recommendation !== "string" || !d.recommendation.trim()))) return "invalid"; + const options = reviveDecisionOptions(d.options); + if (options === "invalid") return "invalid"; + if (options !== undefined && d.recommendation !== undefined + && !options.some((option) => option.trim() === (d.recommendation as string).trim())) return "invalid"; if (d.status === "open") { if (d.answer !== undefined || d.decidedAt !== undefined) return "invalid"; decisions.push({ id: d.id, question: d.question, status: "open", askedAt: d.askedAt, - ...(d.recommendation === undefined ? {} : { recommendation: d.recommendation as string }) }); + ...(d.recommendation === undefined ? {} : { recommendation: d.recommendation as string }), + ...(options === undefined ? {} : { options }) }); } else if (d.status === "decided") { if (typeof d.answer !== "string" || !d.answer.trim() || !validIsoTime(d.decidedAt)) return "invalid"; decisions.push({ id: d.id, question: d.question, status: "decided", answer: d.answer, askedAt: d.askedAt, decidedAt: d.decidedAt, - ...(d.recommendation === undefined ? {} : { recommendation: d.recommendation as string }) }); + ...(d.recommendation === undefined ? {} : { recommendation: d.recommendation as string }), + ...(options === undefined ? {} : { options }) }); } else return "invalid"; } return decisions; @@ -1224,7 +1242,7 @@ export type GoalplanLifecycleResult = export function askGoalplanDecision( plan: Goalplan, - input: { id: string; question: string; recommendation?: string; workPhaseIds: string[]; askedAt: string }, + input: { id: string; question: string; recommendation?: string; options?: string[]; workPhaseIds: string[]; askedAt: string }, ): GoalplanLifecycleResult { const id = input.id.trim(); const question = input.question.trim(); @@ -1233,6 +1251,16 @@ export function askGoalplanDecision( if (!LIFECYCLE_ID_RE.test(id)) return { kind: "rejected", reason: "decision id must be a short lowercase id, e.g. dec-1" }; if (!question) return { kind: "rejected", reason: "decision question must not be empty" }; if (input.recommendation !== undefined && !recommendation) return { kind: "rejected", reason: "decision recommendation must not be empty" }; + const options = input.options?.map((option) => option.trim()); + if (options !== undefined) { + if (options.length === 0) return { kind: "rejected", reason: "decision options must not be empty" }; + if (options.some((option) => !option)) return { kind: "rejected", reason: "decision options must be non-empty text" }; + const repeated = options.find((option, index) => options.indexOf(option) !== index); + if (repeated !== undefined) return { kind: "rejected", reason: `duplicate decision option '${repeated}'` }; + if (recommendation !== undefined && !options.includes(recommendation)) { + return { kind: "rejected", reason: "decision recommendation must be one of the options" }; + } + } if (!validIsoTime(input.askedAt)) return { kind: "rejected", reason: "decision askedAt must be an ISO timestamp" }; if (plan.decisions?.some((decision) => decision.id === id)) return { kind: "rejected", reason: `decision '${id}' is already in this plan` }; const duplicate = plan.decisions?.find((decision) => decision.status === "open" && decision.question.trim() === question); @@ -1248,7 +1276,8 @@ export function askGoalplanDecision( } } const decision: GoalplanDecision = { id, question, status: "open", askedAt: input.askedAt, - ...(recommendation === undefined ? {} : { recommendation }) }; + ...(recommendation === undefined ? {} : { recommendation }), + ...(options === undefined ? {} : { options }) }; const next: Goalplan = { ...plan, decisions: [...(plan.decisions ?? []), decision], workPhases: plan.workPhases.map((wp) => workPhaseIds.includes(wp.id) ? { ...wp, awaitsDecision: [...(wp.awaitsDecision ?? []), id] } : wp) }; diff --git a/plugins/codexclaw/components/pabcd-state/test/goalplan-public-surface.test.ts b/plugins/codexclaw/components/pabcd-state/test/goalplan-public-surface.test.ts index f7741eb9..4ffb986e 100644 --- a/plugins/codexclaw/components/pabcd-state/test/goalplan-public-surface.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/goalplan-public-surface.test.ts @@ -7,7 +7,7 @@ import { dirname, join, resolve } from "node:path"; import { spawnSync } from "node:child_process"; import { fileURLToPath } from "node:url"; import { - advanceWorkPhase, buildGoalplan, goalplanDir, readGoalplan, readyTasks, + advanceWorkPhase, askGoalplanDecision, buildGoalplan, goalplanDir, readGoalplan, readyTasks, readyWorkPhases, writeGoalplan, type Goalplan, } from "../src/goalplan.ts"; import { @@ -778,3 +778,109 @@ test("ready rejects dangling and duplicate decision references", () => { writeGoalplan(cwd, plan); assert.match(cli(cwd, ["ready", "--session", session]).output, /duplicate decision id/); }); + + +// #262 follow-up (issue train 0930, devlog/_plan/260930_issue_train/030): decision options + +test("ask records options and ready/show expose them", () => { + const plan = fixture(); + const { cwd, session } = workspace(plan); + const asked = cli(cwd, ["ask", "--session", session, "--id", "dec-1", "--question", "Choose API", + "--option", "A", "--option", "B", "--recommendation", "A", "--work-phase", "wp-live"]); + assert.equal(asked.code, 0, asked.output); + assert.deepEqual(readGoalplan(cwd, plan.slug)!.decisions?.[0]?.options, ["A", "B"]); + const data = JSON.parse(cli(cwd, ["ready", "--session", session, "--json"]).output); + assert.deepEqual(data.openDecisions[0].options, ["A", "B"]); + assert.match(cli(cwd, ["show", "--session", session]).output, /options: A \| B \(recommended: A\)/); + assert.match(renderGoalplanHelp(), /\[--option \]\.\.\./); +}); + +test("ask rejects a recommendation outside the options without a write", () => { + const plan = fixture(); + const { cwd, session } = workspace(plan); + const before = planText(cwd, plan.slug), ledger = ledgerText(cwd, plan.slug); + const res = cli(cwd, ["ask", "--session", session, "--id", "dec-1", "--question", "Choose API", + "--option", "A", "--option", "B", "--recommendation", "C"]); + assert.equal(res.code, 1); + assert.match(res.output, /must be one of the options/); + assert.equal(planText(cwd, plan.slug), before); + assert.equal(ledgerText(cwd, plan.slug), ledger); +}); + +test("ask rejects blank and repeated options at parse time", () => { + const plan = fixture(); + const { cwd, session } = workspace(plan); + const before = planText(cwd, plan.slug); + const blank = parseGoalplanCliArgs(["ask", "--session", session, "--id", "dec-1", "--question", "Q", "--option", " "], cwd); + assert.match((blank as { error: string }).error, /--option requires one non-empty value/); + const repeated = parseGoalplanCliArgs(["ask", "--session", session, "--id", "dec-1", "--question", "Q", "--option", "A", "--option", " A"], cwd); + assert.match((repeated as { error: string }).error, /--option must not repeat 'A'/); + const misplaced = parseGoalplanCliArgs(["decide", "--session", session, "--id", "dec-1", "--answer", "x", "--option", "A"], cwd); + assert.equal("error" in misplaced, true); + assert.equal(planText(cwd, plan.slug), before); +}); + +test("ask without --option stores no options key", () => { + const plan = fixture(); + const { cwd, session } = workspace(plan); + assert.equal(cli(cwd, ["ask", "--session", session, "--id", "dec-1", "--question", "Choose API"]).code, 0); + const raw = JSON.parse(planText(cwd, plan.slug)); + assert.equal("options" in raw.decisions[0], false); +}); + +test("decide keeps options and accepts a free-form answer", () => { + const plan = fixture(); + const { cwd, session } = workspace(plan); + cli(cwd, ["ask", "--session", session, "--id", "dec-1", "--question", "Choose API", "--option", "A", "--option", "B"]); + const decided = cli(cwd, ["decide", "--session", session, "--id", "dec-1", "--answer", "something else"]); + assert.equal(decided.code, 0, decided.output); + const back = readGoalplan(cwd, plan.slug)!.decisions![0]; + assert.equal(back.status, "decided"); + assert.equal(back.answer, "something else"); + assert.deepEqual(back.options, ["A", "B"]); +}); + +test("reviver fails closed on malformed options", () => { + const plan = fixture(); + const { cwd, session } = workspace(plan); + assert.equal(cli(cwd, ["ask", "--session", session, "--id", "dec-1", "--question", "Choose API", "--option", "A", "--recommendation", "A"]).code, 0); + const good = JSON.parse(planText(cwd, plan.slug)); + const path = join(goalplanDir(cwd, plan.slug), "goalplan.json"); + for (const mutate of [ + (d: any) => { d.options = {}; }, + (d: any) => { d.options = []; }, + (d: any) => { d.options = [" "]; }, + (d: any) => { d.options = ["A", "A "]; }, + (d: any) => { d.options = [1]; }, + (d: any) => { d.options = ["B"]; }, + ]) { + const bad = JSON.parse(JSON.stringify(good)); + mutate(bad.decisions[0]); + writeFileSync(path, JSON.stringify(bad, null, 2)); + assert.equal(readGoalplan(cwd, plan.slug), null, JSON.stringify(bad.decisions[0].options)); + assert.match(cli(cwd, ["show", "--session", session]).output, /field 'decisions'/); + } + const padded = JSON.parse(JSON.stringify(good)); + padded.decisions[0].options = [" A "]; + writeFileSync(path, JSON.stringify(padded, null, 2)); + assert.deepEqual(readGoalplan(cwd, plan.slug)!.decisions![0].options, [" A "]); +}); + +test("askGoalplanDecision rejects empty, blank and repeated options (library)", () => { + const plan = fixture(); + const base = { id: "dec-1", question: "Choose API", workPhaseIds: [], askedAt: "2026-09-30T00:00:00.000Z" }; + const reason = (options: string[]) => { + const res = askGoalplanDecision(plan, { ...base, options }); + assert.equal(res.kind, "rejected"); + return (res as { reason: string }).reason; + }; + assert.match(reason([]), /must not be empty/); + assert.match(reason([" "]), /non-empty text/); + assert.match(reason(["A", " A"]), /duplicate decision option 'A'/); + const ok = askGoalplanDecision(plan, { ...base, options: [" A ", "B"], recommendation: " A" }); + assert.equal(ok.kind, "changed"); + const stored = (ok as { plan: Goalplan }).plan.decisions![0]; + assert.deepEqual(stored.options, ["A", "B"]); + assert.equal(stored.recommendation, "A"); +}); + diff --git a/plugins/codexclaw/skills/dev/references/async-questions.md b/plugins/codexclaw/skills/dev/references/async-questions.md index 0d38e95b..e31acf4d 100644 --- a/plugins/codexclaw/skills/dev/references/async-questions.md +++ b/plugins/codexclaw/skills/dev/references/async-questions.md @@ -53,7 +53,7 @@ earlier answer must wait. Do not inherit the blocking tool's three-question limi within the authorized scope. A missing reply never blocks completion of optional work. Required input or approval remains a real dependency: silence and preselection are never consent; only the dependent action stays pending. -For a bound goalplan, after the host confirms a question was submitted, run `cxc loop ask --session --id --question [--recommendation ] [--work-phase ]...` to keep it in the existing plan. Record each dependent work phase. Check `cxc loop show` before sending a similar question after compaction. When the user answers, run `cxc loop decide --session --id --answer ` and resume only newly runnable phases. `ask` does not deliver a question; CLI success is no proof of host submission. Link a phase only while its action truly requires the reply. +For a bound goalplan, after the host confirms a question was submitted, run `cxc loop ask --session --id --question [--recommendation ] [--option ]... [--work-phase ]...` (the offered options, recommended first) to keep it in the existing plan. Record each dependent work phase. Check `cxc loop show` before sending a similar question after compaction. When the user answers, run `cxc loop decide --session --id --answer ` and resume only newly runnable phases. `ask` does not deliver a question; CLI success is no proof of host submission. Link a phase only while its action truly requires the reply. 4. A later user message supplies the answer. Match it to the pending decision, update assumptions and affected work, and retain the original objective unless diff --git a/plugins/codexclaw/skills/loop/references/durable-goalplan.md b/plugins/codexclaw/skills/loop/references/durable-goalplan.md index 822d998f..a638bf7d 100644 --- a/plugins/codexclaw/skills/loop/references/durable-goalplan.md +++ b/plugins/codexclaw/skills/loop/references/durable-goalplan.md @@ -60,7 +60,7 @@ This is the on-disk shape under `.codexclaw/goalplans//goalplan.json` Task ids and task dependency references are phase-local: `task.dependsOn` names existing task ids in the same work phase, never a task in another phase. A done task carries a non-empty `outcome`; a pending task has no outcome. -- Optional `decisions[]` — each `{ id, question, recommendation?, status: open|decided, answer?, askedAt, decidedAt? }`. Open decisions have no answer or decidedAt; decided decisions require both. Decision ids are short lowercase ids. Absent and empty arrays remain distinct on disk, as do absent and empty `awaitsDecision` arrays. Old plans acquire neither field on read/write. Only linked pending or in-progress phases wait; an unrelated open decision does not pause the goal. +- Optional `decisions[]` — each `{ id, question, recommendation?, options?, status: open|decided, answer?, askedAt, decidedAt? }`. When `options` is present it is a non-empty list of distinct entries and `recommendation` must be one of them; the answer stays free text because the host always offers a free-form reply. Open decisions have no answer or decidedAt; decided decisions require both. Decision ids are short lowercase ids. Absent and empty arrays remain distinct on disk, as do absent and empty `awaitsDecision` arrays. Old plans acquire neither field on read/write. Only linked pending or in-progress phases wait; an unrelated open decision does not pause the goal. - `criteria[]` — each `{ id, scenario, surface, presented?, expectedEvidence, capturedEvidence, status: open|met }`. `scenario` is the `--criterion` text and `surface` is one of `logic` (default), `web`, `tui` or `desktop`, set by `add-criterion --surface` on a session-bound plan @@ -98,7 +98,7 @@ This is the on-disk shape under `.codexclaw/goalplans//goalplan.json` - `cxc loop ready (--slug | --objective | --session ) [--json] [--cwd ]` - `cxc loop add-task --session --work-phase --id --title [--depends-on ]... [--cwd ]` - `cxc loop complete-task --session --work-phase --id --outcome [--cwd ]` -- `cxc loop ask --session --id --question [--recommendation ] [--work-phase ]... [--cwd ]` — record a question after sending it through the host. It never sends a message. Name each dependent phase. +- `cxc loop ask --session --id --question [--recommendation ] [--option ]... [--work-phase ]... [--cwd ]` — record a question after sending it through the host. It never sends a message. Name each dependent phase. - `cxc loop decide --session --id --answer [--cwd ]` — record the user reply. This changes only the decision record; phase status and blockedReason stay as they were. - `cxc loop meet-criterion --session --id --evidence [--cwd ]` — `--id` takes a generated `c-N` id; read it from `cxc loop show` or the goalplan file. From 0c0ae34f1809251d012f2a5a11630fe41b91d23f Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Wed, 30 Sep 2026 03:13:48 +0900 Subject: [PATCH 87/90] docs(plan): wp5 B record; badges 3737 --- README.ko.md | 2 +- README.md | 2 +- README.zh.md | 2 +- devlog/_plan/260930_issue_train/030_wp5_decision_options.md | 4 ++++ 4 files changed, 7 insertions(+), 3 deletions(-) diff --git a/README.ko.md b/README.ko.md index 07fb562a..41048f13 100644 --- a/README.ko.md +++ b/README.ko.md @@ -13,7 +13,7 @@

CI - 3,730 tests + 3,737 tests 29 skills 31 hooks Documentation diff --git a/README.md b/README.md index 5a8095c9..9b078245 100644 --- a/README.md +++ b/README.md @@ -13,7 +13,7 @@

CI - 3,730 tests + 3,737 tests 29 skills 31 hooks Documentation diff --git a/README.zh.md b/README.zh.md index 426029e9..598b3d30 100644 --- a/README.zh.md +++ b/README.zh.md @@ -13,7 +13,7 @@

CI - 3,730 tests + 3,737 tests 29 skills 31 hooks Documentation diff --git a/devlog/_plan/260930_issue_train/030_wp5_decision_options.md b/devlog/_plan/260930_issue_train/030_wp5_decision_options.md index 06ca5864..3a513acb 100644 --- a/devlog/_plan/260930_issue_train/030_wp5_decision_options.md +++ b/devlog/_plan/260930_issue_train/030_wp5_decision_options.md @@ -170,3 +170,7 @@ Continuity (LOOP-CONTINUITY-01), quoting the wp3 D summary in 020: "#275 is addr Re-checked on `codex/issue-train-0930-wp5` from `origin/dev` `58a8a174`: `git diff --stat 659de59b..HEAD -- plugins/codexclaw/components/pabcd-state plugins/codexclaw/skills/dev/references/async-questions.md` is empty, so every `goalplan.ts` and `goalplan-cli.ts` anchor above holds. As planned, wp3's three-line edit moved the `durable-goalplan.md` anchors: the `decisions[]` schema line is now `:63` (was `:60`) and the `ask` synopsis is `:101` (was `:98`). `async-questions.md:56` is unchanged. `renderGoalplanHelp` (`goalplan-cli.ts:702`) prints the verb usage strings, so it picks up the new `ask` usage without a separate edit. No design decision changes, so no architect recheck. Amendment (architect's optional D26 note): add one Notes line to `renderGoalplanHelp` after the `decide` note (`goalplan-cli.ts:724`): `" Repeat --option once per offered option; the recommendation must be one of them, and the answer stays free text."` No test pins the help Notes text (checked with `rg 'Record the user' plugins/codexclaw/components/pabcd-state/test`). + +## wp5 B record (2026-09-30) + +Built at `355afde3` per the file map (1a-1d, 2a-2h, docs 5), plus seven tests (six planned rows and the library test). Red check: with `goalplan.ts` and `goalplan-cli.ts` restored from `58a8a174` in place (`git show 58a8a174: > `, then `git checkout HEAD -- `), the committed test file runs 41 tests, 35 pass, 6 fail: every new behavior test; the seventh ("ask without --option stores no options key") is a compatibility test that passes on both sources by design. New source: 41/41. Badges move to 3737. From 7e4b90a3518fafb1fadd95cb43a3ed86fab4ed6c Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Wed, 30 Sep 2026 03:17:57 +0900 Subject: [PATCH 88/90] test(goalplan): observe option rendering branches; async-questions wording --- .../260930_issue_train/030_wp5_decision_options.md | 6 ++++++ .../pabcd-state/test/goalplan-public-surface.test.ts | 10 +++++++++- .../codexclaw/skills/dev/references/async-questions.md | 2 +- 3 files changed, 16 insertions(+), 2 deletions(-) diff --git a/devlog/_plan/260930_issue_train/030_wp5_decision_options.md b/devlog/_plan/260930_issue_train/030_wp5_decision_options.md index 3a513acb..698ba7a1 100644 --- a/devlog/_plan/260930_issue_train/030_wp5_decision_options.md +++ b/devlog/_plan/260930_issue_train/030_wp5_decision_options.md @@ -174,3 +174,9 @@ Amendment (architect's optional D26 note): add one Notes line to `renderGoalplan ## wp5 B record (2026-09-30) Built at `355afde3` per the file map (1a-1d, 2a-2h, docs 5), plus seven tests (six planned rows and the library test). Red check: with `goalplan.ts` and `goalplan-cli.ts` restored from `58a8a174` in place (`git show 58a8a174: > `, then `git checkout HEAD -- `), the committed test file runs 41 tests, 35 pass, 6 fail: every new behavior test; the seventh ("ask without --option stores no options key") is a compatibility test that passes on both sources by design. New source: 41/41. Badges move to 3737. + +## wp5 C record (2026-09-30) + +C round 1 on `0c0ae34f`: fresh implementation reviewer `01a0ee5f-7c4f` PASS (four Low findings), initiative verifier `01a0ee5f-7d90` GO-WITH-FIXES (4, all procedural: finished gate, hosted CI, independent review verdict, goalplan records); the initiative verifier also reproduced the red/green counts in an isolated `git archive`-style export (35/6 red, 41/0 green). C gate on `0c0ae34f`: 3737 tests, 0 failures, inventory, gate, smoke, hook diff 0. Folded: assertions (no new tests, count stays 3737) for `show` with options and no recommendation, `ready --json` without options, the exact `unknown flag '--option` error, and the `--option=value` form; `async-questions.md:56` now says "recommended first by convention". + +Residual for the wp4 CHANGELOG: builds older than this one rebuild decisions field by field and drop `options` if they rewrite the plan (no schema-version bump signals the key). Criterion c-10 is linked to wp5 by its text (work-phase `criteriaIds` are empty in this goalplan); #262 entered through the objective's "worthwhile improvements among the open issues" outcome, not its enumerated Scope IN list. diff --git a/plugins/codexclaw/components/pabcd-state/test/goalplan-public-surface.test.ts b/plugins/codexclaw/components/pabcd-state/test/goalplan-public-surface.test.ts index 4ffb986e..19cc8248 100644 --- a/plugins/codexclaw/components/pabcd-state/test/goalplan-public-surface.test.ts +++ b/plugins/codexclaw/components/pabcd-state/test/goalplan-public-surface.test.ts @@ -816,7 +816,9 @@ test("ask rejects blank and repeated options at parse time", () => { const repeated = parseGoalplanCliArgs(["ask", "--session", session, "--id", "dec-1", "--question", "Q", "--option", "A", "--option", " A"], cwd); assert.match((repeated as { error: string }).error, /--option must not repeat 'A'/); const misplaced = parseGoalplanCliArgs(["decide", "--session", session, "--id", "dec-1", "--answer", "x", "--option", "A"], cwd); - assert.equal("error" in misplaced, true); + assert.match((misplaced as { error: string }).error, /unknown flag '--option/); + const equalsForm = parseGoalplanCliArgs(["ask", "--session", session, "--id", "dec-1", "--question", "Q", "--option=A", "--option=--x"], cwd); + assert.deepEqual((equalsForm as GoalplanCliArgs).options, ["A", "--x"]); assert.equal(planText(cwd, plan.slug), before); }); @@ -826,6 +828,8 @@ test("ask without --option stores no options key", () => { assert.equal(cli(cwd, ["ask", "--session", session, "--id", "dec-1", "--question", "Choose API"]).code, 0); const raw = JSON.parse(planText(cwd, plan.slug)); assert.equal("options" in raw.decisions[0], false); + const data = JSON.parse(cli(cwd, ["ready", "--session", session, "--json"]).output); + assert.equal("options" in data.openDecisions[0], false); }); test("decide keeps options and accepts a free-form answer", () => { @@ -838,6 +842,10 @@ test("decide keeps options and accepts a free-form answer", () => { assert.equal(back.status, "decided"); assert.equal(back.answer, "something else"); assert.deepEqual(back.options, ["A", "B"]); + cli(cwd, ["ask", "--session", session, "--id", "dec-2", "--question", "Pick a region", "--option", "eu", "--option", "us"]); + const shown = cli(cwd, ["show", "--session", session]).output; + assert.match(shown, /options: eu \| us$/m); + assert.doesNotMatch(shown, /recommended: undefined/); }); test("reviver fails closed on malformed options", () => { diff --git a/plugins/codexclaw/skills/dev/references/async-questions.md b/plugins/codexclaw/skills/dev/references/async-questions.md index e31acf4d..a14050f2 100644 --- a/plugins/codexclaw/skills/dev/references/async-questions.md +++ b/plugins/codexclaw/skills/dev/references/async-questions.md @@ -53,7 +53,7 @@ earlier answer must wait. Do not inherit the blocking tool's three-question limi within the authorized scope. A missing reply never blocks completion of optional work. Required input or approval remains a real dependency: silence and preselection are never consent; only the dependent action stays pending. -For a bound goalplan, after the host confirms a question was submitted, run `cxc loop ask --session --id --question [--recommendation ] [--option ]... [--work-phase ]...` (the offered options, recommended first) to keep it in the existing plan. Record each dependent work phase. Check `cxc loop show` before sending a similar question after compaction. When the user answers, run `cxc loop decide --session --id --answer ` and resume only newly runnable phases. `ask` does not deliver a question; CLI success is no proof of host submission. Link a phase only while its action truly requires the reply. +For a bound goalplan, after the host confirms a question was submitted, run `cxc loop ask --session --id --question [--recommendation ] [--option ]... [--work-phase ]...` (list the offered options, recommended first by convention; the recommendation must be one of them) to keep it in the existing plan. Record each dependent work phase. Check `cxc loop show` before sending a similar question after compaction. When the user answers, run `cxc loop decide --session --id --answer ` and resume only newly runnable phases. `ask` does not deliver a question; CLI success is no proof of host submission. Link a phase only while its action truly requires the reply. 4. A later user message supplies the answer. Match it to the pending decision, update assumptions and affected work, and retain the original objective unless From 419ab7f15096bc2ebb4cd2a2482ed37641267a71 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Wed, 30 Sep 2026 03:29:58 +0900 Subject: [PATCH 89/90] docs(plan): wp5 D summary; wp4 P revalidation --- .../030_wp5_decision_options.md | 6 ++++ .../260930_issue_train/040_wp4_delivery.md | 28 +++++++++++++++++++ 2 files changed, 34 insertions(+) diff --git a/devlog/_plan/260930_issue_train/030_wp5_decision_options.md b/devlog/_plan/260930_issue_train/030_wp5_decision_options.md index 698ba7a1..99662820 100644 --- a/devlog/_plan/260930_issue_train/030_wp5_decision_options.md +++ b/devlog/_plan/260930_issue_train/030_wp5_decision_options.md @@ -180,3 +180,9 @@ Built at `355afde3` per the file map (1a-1d, 2a-2h, docs 5), plus seven tests (s C round 1 on `0c0ae34f`: fresh implementation reviewer `01a0ee5f-7c4f` PASS (four Low findings), initiative verifier `01a0ee5f-7d90` GO-WITH-FIXES (4, all procedural: finished gate, hosted CI, independent review verdict, goalplan records); the initiative verifier also reproduced the red/green counts in an isolated `git archive`-style export (35/6 red, 41/0 green). C gate on `0c0ae34f`: 3737 tests, 0 failures, inventory, gate, smoke, hook diff 0. Folded: assertions (no new tests, count stays 3737) for `show` with options and no recommendation, `ready --json` without options, the exact `unknown flag '--option` error, and the `--option=value` form; `async-questions.md:56` now says "recommended first by convention". Residual for the wp4 CHANGELOG: builds older than this one rebuild decisions field by field and drop `options` if they rewrite the plan (no schema-version bump signals the key). Criterion c-10 is linked to wp5 by its text (work-phase `criteriaIds` are empty in this goalplan); #262 entered through the objective's "worthwhile improvements among the open issues" outcome, not its enumerated Scope IN list. + +## wp5 D summary (2026-09-30) + +Conclusion: the #262 options half is merged into `dev` through PR #280 (head `7e4b90a3`, 14/14 checks, merge `99c9df6a`); #262 stays open for `withdrawn`. Evidence: red 6/41 on the `58a8a174` source (reproduced independently), 41/41 on the new source, C gate on the final head (3737 tests, 0 failures). Next: wp4 delivers per 040. + +What did not go well: the first test set left two rendering branches unobserved and asserted a parse error too loosely; the red check swapped tracked files in place in a shared tree, which worked but is riskier than an export. The downgrade residual (older builds drop `options` on rewrite) was found only at C. Evidence that the direction is wrong: users need `withdrawn` or answer-to-option linking more than option lists, which would show up as `decide` answers that repeat an option verbatim. diff --git a/devlog/_plan/260930_issue_train/040_wp4_delivery.md b/devlog/_plan/260930_issue_train/040_wp4_delivery.md index bc7696c1..a23b23ee 100644 --- a/devlog/_plan/260930_issue_train/040_wp4_delivery.md +++ b/devlog/_plan/260930_issue_train/040_wp4_delivery.md @@ -38,3 +38,31 @@ The installed plugin cache and remote hosts are not updated by this train (goal ## Acceptance All goalplan criteria met with captured evidence; `cxc loop validate` passes; v0.2.40 is the latest release and its assets verify. + +## wp4 P revalidation and executable amendment (2026-09-30) + +Continuity (LOOP-CONTINUITY-01), quoting the wp5 D summary in 030: "the #262 options half is merged ... Next: wp4 delivers per 040." State at entry: `origin/dev` = `99c9df6a` with PRs #278 (`069a7d0e`), #279 (`58a8a174`), #280 (`99c9df6a`) merged after 14/14 checks on their heads; `main` = `8e6aa800` (v0.2.39); no v0.2.40 tag or release exists. Branch `codex/release-0240` from `99c9df6a`. No architect consultation: this phase makes no design decisions (the 0927 train's wp5 precedent); the A reviewers cover the steps. + +Resource bounds (disclosed gap): the release is C4 and the initiative's loop-engineering rule asks for a token and wall-clock bound; the user authorized push, merge to `dev` and `main`, and release on 2026-09-30 without stating one, so none is invented. Stop conditions instead: any red check on an exact head, a release dry run that is not READY, or an asset mismatch halts delivery with the state reported. + +### Version edits (re-verified with the `rg` in step 2 at `99c9df6a`) + +`0.2.39` -> `0.2.40` in `package.json`, `cli/package.json`, `plugins/codexclaw/gui/package.json`, the nine `plugins/codexclaw/components/*/package.json`, and the 13 `"version": "0.2.39"` entries in `package-lock.json` (root and workspace entries). `plugins/codexclaw/.codex-plugin/plugin.json`: `"version": "0.2.40+codex."`. `inventory.json` component versions and README badges via `inventory.mjs --write --tests 3737`. `pabcd-state/test/hook.test.ts:181` contains `0.2.39` only inside a fixture cache path; it stays. + +### CHANGELOG diff + +`## [Unreleased]` becomes `## [0.2.40] - 2026-09-30`, a fresh empty `## [Unreleased]` goes above it, and these lines join the existing sections: + +- Added: "Dispatch packets can declare each verifier's write effects (`verifierEffects`), and a pure `verifierPreflight(packet)` reports which verifiers need an isolated copy: under a shared-read packet only a verifier declared read-only runs in the shared tree. Nothing executes a command (#277)." +- Added: "Interview assumptions carry their source, confidence, consequence if wrong and a status (`proposed`, `open`, `user_confirmed`, `user_rejected`); confirmed and rejected entries need an answer reference, and the plan keeps open assumptions apart from confirmed requirements and rejected ones (INTERVIEW-ASSUME-01, guidance only, #275)." +- Changed (extend the existing #262 bullet): "`cxc loop ask` also takes a repeatable `--option `; when options are given the recommendation must be one of them, the answer stays free text, and `ready --json` and `show` list them. Builds older than 0.2.40 drop `options` if they rewrite such a plan." +- Fixed: "A dispatch receipt satisfies its packet only when every required verifier command has a matching result with exit 0 and, when commands are required, no result names another command. Receipts can report `verifierResults[]`; a single legacy `verifierResult` for a multi-command packet reports incomplete. `validateReceipt` now checks the result shapes and `validatePacket` rejects blank or non-string verifier commands; a receipt whose one result names a different command, even cosmetically, no longer satisfies (#276)." + +### Issue comments (wording) + +- #276, #277, #275: "Fixed in #278/#279 (merged into `dev` as ) and released in v0.2.40." closed as completed. +- #262: "Options shipped in #280 (`ask --option`, recommendation must be one of them, answers stay free text). `withdrawn` still needs a decision on how a withdrawn question releases its linked phases, so this stays open." +- Not planned (#209, #213, #247, #258, #259, #263, #264, #265, #266, #267, #268): the reason line from 001 and "Closing as not planned in the 2026-09-30 issue train: codexclaw is keeping its hook surface small, and this needs ." plus the link to 001 on `dev`. +- Kept open (#255, #256, #257, #260, #273, #274): the reason line from 001 and the link. + +Order: issue comments and closes run after the release so "released in v0.2.40" is true. From 1f02cfd40d49b1a8752a599b90436f58f4768651 Mon Sep 17 00:00:00 2001 From: bitkyc08-arch Date: Wed, 30 Sep 2026 03:32:41 +0900 Subject: [PATCH 90/90] release: codexclaw 0.2.40 --- CHANGELOG.md | 16 +++++++++++- cli/package.json | 2 +- package-lock.json | 26 +++++++++---------- package.json | 2 +- plugins/codexclaw/.codex-plugin/plugin.json | 2 +- .../codexclaw/components/bg-wake/package.json | 2 +- .../components/config-guard/package.json | 2 +- .../codexclaw/components/cxc-ops/package.json | 2 +- .../components/messenger-bridge/package.json | 2 +- .../components/pabcd-state/package.json | 2 +- .../components/provider-bridge/package.json | 2 +- .../codexclaw/components/recall/package.json | 2 +- .../components/skill-search/package.json | 2 +- .../components/subagent-config/package.json | 2 +- plugins/codexclaw/gui/package.json | 2 +- plugins/codexclaw/inventory.json | 22 ++++++++-------- 16 files changed, 52 insertions(+), 38 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 0c6d0248..f14ea3f4 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,24 +6,38 @@ All notable changes to codexclaw are documented here. The format follows ## [Unreleased] +## [0.2.40] - 2026-09-29 + ### Added - Codex Desktop sometimes starts threads created by `create_thread` with on-request approvals even when the user's config is full access (openai/codex #33282). A SessionStart advisory now tells the model and the user when an agent-created thread starts that way. An opt-in PermissionRequest hook (`permissions.agentCreatedThreadAutoAllow: true` in `~/.codexclaw/config.json`, off by default) answers those threads' approval prompts, including one-time network requests, only when the user's top-level `config.toml` sets `approval_policy = "never"` and `sandbox_mode = "danger-full-access"`; it never changes the thread's sandbox, never denies, and ignores project-local config. Two new hooks (31 total) need trust approval after upgrade. - Dispatch guidance: for bounded worktree lanes a full-access coordinator can create a managed worktree and hand a subagent that path as its shell `workdir`, which keeps the coordinator's permission; workers may keep an optional `PROGRESS.md` checkpoint so a replacement can resume from files (#265, guidance only). - `CODEXCLAW_PABCD=off` (or `on`) and project `codexclaw.json` `{"pabcd": {"enabled": false}}` turn the PABCD hook policy off while keeping the worktree, memory-write, automation-ownership and apply_patch lint guards and recall active. A recognized environment value wins over the project file in both directions (#252). - When codexclaw creates a project's `.codexclaw` folder, it also writes `.codexclaw/.gitignore` so session state, ledgers and evidence stay out of git; user-authored `rules/*.md` stay committable unless an ancestor ignore rule hides the folder. Existing `.codexclaw` folders are never modified. Lazy creation of session state is deferred (#255, partial). +- Dispatch packets can declare each verifier's write effects (`verifierEffects`), and a pure `verifierPreflight(packet)` reports which verifiers need an isolated copy: under a shared-read packet only a verifier declared read-only runs in the shared tree. Nothing executes a command (#277). +- Interview assumptions carry their source, confidence, consequence if wrong and a status (`proposed`, `open`, `user_confirmed`, `user_rejected`); confirmed and rejected entries need an answer reference, and the plan keeps open assumptions apart from confirmed requirements and rejected ones (INTERVIEW-ASSUME-01, guidance only, #275). ### Changed -- Goalplans can record pending user decisions: `cxc loop ask --session --id --question [--recommendation ] [--work-phase ]...` links a question the agent already asked to the phases that wait on it, and `cxc loop decide --session --id --answer ` records the answer. Linked phases are not runnable while the decision is open; unrelated phases stay ready. When every remaining phase and unmet criterion waits on an open decision, the Stop hook lets an IDLE turn end instead of asking to start another phase; the goal stays active and cannot be completed early. Old plans load unchanged (#262). +- Goalplans can record pending user decisions: `cxc loop ask --session --id --question [--recommendation ] [--work-phase ]...` links a question the agent already asked to the phases that wait on it, and `cxc loop decide --session --id --answer ` records the answer. Linked phases are not runnable while the decision is open; unrelated phases stay ready. When every remaining phase and unmet criterion waits on an open decision, the Stop hook lets an IDLE turn end instead of asking to start another phase; the goal stays active and cannot be completed early. Old plans load unchanged (#262). `cxc loop ask` also takes a repeatable `--option `; when options are given the recommendation must be one of them, the answer stays free text, and `ready --json` and `show` list them. - The absolute Stop continuation cap (24) now counts per genuine user turn instead of per session, and the release prints one notice per turn (#254). ### Fixed +- A dispatch receipt satisfies its packet only when every required verifier command has a matching result with exit 0 and, when commands are required, no result names another command. Receipts can report `verifierResults[]`; a single legacy `verifierResult` for a multi-command packet reports incomplete (#276). - Ordinary words (for example "interview", "keep going until", "끝까지 진행해", quoted or fenced examples) no longer inject PABCD phase directives or arm the loop; hints need an explicit codexclaw request such as `cxc-pabcd` or `cxc-loop` (#250). - The SubagentStop evidence gate no longer blocks Codex's built-in `worker` outside an active PABCD build or check cycle; registered `executor` stays gated while PABCD is on (#251). - An active native goal without a bound goalplan no longer blocks Stop at IDLE (#253). +### Compatibility + +- `receiptSatisfiesPacket` is stricter (#276): a receipt whose one result names a different command than the packet's, even cosmetically (`npm run test` vs `npm test`), no longer satisfies; extra passing checks belong in `commandsRun`. `validateReceipt` now checks the verifier result shapes and `validatePacket` rejects blank or non-string verifier commands. +- Builds older than 0.2.40 drop a goalplan decision's `options` if they rewrite the plan; no schema-version bump signals the new key. +- The two hooks added in this release (31 total) need trust approval after upgrade. + +### Verification + +- 3737 tests, 0 failures (`npm test`); `gate.mjs`, inventory and `platform-smoke.mjs` pass. Hosted CI and the packed-install lifecycle passed on every merged pull request (#269-#272, #278-#280). ## [0.2.39] - 2026-09-24 diff --git a/cli/package.json b/cli/package.json index e10c2d4a..d0fec216 100644 --- a/cli/package.json +++ b/cli/package.json @@ -1,6 +1,6 @@ { "name": "@codexclaw/cli", - "version": "0.2.39", + "version": "0.2.40", "private": true, "type": "module", "description": "codexclaw CLI \u2014 status, subagent config, provider toggle, GUI launcher.", diff --git a/package-lock.json b/package-lock.json index 21701540..bd3e0112 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "codexclaw", - "version": "0.2.39", + "version": "0.2.40", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "codexclaw", - "version": "0.2.39", + "version": "0.2.40", "license": "MIT", "workspaces": [ "plugins/codexclaw/components/*", @@ -20,7 +20,7 @@ }, "cli": { "name": "@codexclaw/cli", - "version": "0.2.39", + "version": "0.2.40", "bin": { "codexclaw": "bin/codexclaw.mjs" } @@ -1953,43 +1953,43 @@ }, "plugins/codexclaw/components/bg-wake": { "name": "@codexclaw/bg-wake", - "version": "0.2.39" + "version": "0.2.40" }, "plugins/codexclaw/components/config-guard": { "name": "@codexclaw/config-guard", - "version": "0.2.39" + "version": "0.2.40" }, "plugins/codexclaw/components/cxc-ops": { "name": "@codexclaw/cxc-ops", - "version": "0.2.39" + "version": "0.2.40" }, "plugins/codexclaw/components/messenger-bridge": { "name": "@codexclaw/messenger-bridge", - "version": "0.2.39" + "version": "0.2.40" }, "plugins/codexclaw/components/pabcd-state": { "name": "@codexclaw/pabcd-state", - "version": "0.2.39" + "version": "0.2.40" }, "plugins/codexclaw/components/provider-bridge": { "name": "@codexclaw/provider-bridge", - "version": "0.2.39" + "version": "0.2.40" }, "plugins/codexclaw/components/recall": { "name": "@codexclaw/recall", - "version": "0.2.39" + "version": "0.2.40" }, "plugins/codexclaw/components/skill-search": { "name": "@codexclaw/skill-search", - "version": "0.2.39" + "version": "0.2.40" }, "plugins/codexclaw/components/subagent-config": { "name": "@codexclaw/subagent-config", - "version": "0.2.39" + "version": "0.2.40" }, "plugins/codexclaw/gui": { "name": "@codexclaw/gui", - "version": "0.2.39", + "version": "0.2.40", "dependencies": { "react": "^18.3.1", "react-dom": "^18.3.1" diff --git a/package.json b/package.json index 9cf31582..9f815263 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "codexclaw", - "version": "0.2.39", + "version": "0.2.40", "private": true, "description": "cli-jaw-style dev discipline + multi-model subagents for the OpenAI Codex runtime.", "type": "module", diff --git a/plugins/codexclaw/.codex-plugin/plugin.json b/plugins/codexclaw/.codex-plugin/plugin.json index d48e0fbe..ad49a502 100644 --- a/plugins/codexclaw/.codex-plugin/plugin.json +++ b/plugins/codexclaw/.codex-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "codexclaw", - "version": "0.2.39+codex.20260924082502", + "version": "0.2.40+codex.20260929183231", "description": "cli-jaw-style dev discipline (dev skills + PABCD) and multi-model subagents for the OpenAI Codex runtime, with optional opencodex provider routing.", "author": { "name": "lidge-jun", diff --git a/plugins/codexclaw/components/bg-wake/package.json b/plugins/codexclaw/components/bg-wake/package.json index 9ac8a293..7a34369a 100644 --- a/plugins/codexclaw/components/bg-wake/package.json +++ b/plugins/codexclaw/components/bg-wake/package.json @@ -1,6 +1,6 @@ { "name": "@codexclaw/bg-wake", - "version": "0.2.39", + "version": "0.2.40", "private": true, "type": "module", "description": "Background task registry + completion wake for Codex. Registers detached commands, then wakes the agent through the Stop hook when they finish.", diff --git a/plugins/codexclaw/components/config-guard/package.json b/plugins/codexclaw/components/config-guard/package.json index 2230cb6f..c2399f07 100644 --- a/plugins/codexclaw/components/config-guard/package.json +++ b/plugins/codexclaw/components/config-guard/package.json @@ -1,6 +1,6 @@ { "name": "@codexclaw/config-guard", - "version": "0.2.39", + "version": "0.2.40", "private": true, "type": "module", "description": "Controlled feature-flag activation: enables only codexclaw's declared [features] flags via the official `codex features` CLI, with a revert manifest and backup.", diff --git a/plugins/codexclaw/components/cxc-ops/package.json b/plugins/codexclaw/components/cxc-ops/package.json index 38e952d8..6a7a532c 100644 --- a/plugins/codexclaw/components/cxc-ops/package.json +++ b/plugins/codexclaw/components/cxc-ops/package.json @@ -1,6 +1,6 @@ { "name": "@codexclaw/cxc-ops", - "version": "0.2.39", + "version": "0.2.40", "private": true, "type": "module", "description": "codexclaw ops CLI \u2014 doctor (plugin health), reset (scoped state cleanup).", diff --git a/plugins/codexclaw/components/messenger-bridge/package.json b/plugins/codexclaw/components/messenger-bridge/package.json index 656e39d9..2b5ed058 100644 --- a/plugins/codexclaw/components/messenger-bridge/package.json +++ b/plugins/codexclaw/components/messenger-bridge/package.json @@ -1,6 +1,6 @@ { "name": "@codexclaw/messenger-bridge", - "version": "0.2.39", + "version": "0.2.40", "private": true, "type": "module", "description": "codexclaw messenger bridge \u2014 cxc serve HTTP server + SQLite state substrate (zero third-party deps).", diff --git a/plugins/codexclaw/components/pabcd-state/package.json b/plugins/codexclaw/components/pabcd-state/package.json index 898d370f..9804aa9a 100644 --- a/plugins/codexclaw/components/pabcd-state/package.json +++ b/plugins/codexclaw/components/pabcd-state/package.json @@ -1,6 +1,6 @@ { "name": "@codexclaw/pabcd-state", - "version": "0.2.39", + "version": "0.2.40", "private": true, "type": "module", "description": "IPABCD finite-state machine backed by per-session .codexclaw/sessions/.json + shared ledger.jsonl.", diff --git a/plugins/codexclaw/components/provider-bridge/package.json b/plugins/codexclaw/components/provider-bridge/package.json index e89a28fd..d679a650 100644 --- a/plugins/codexclaw/components/provider-bridge/package.json +++ b/plugins/codexclaw/components/provider-bridge/package.json @@ -1,6 +1,6 @@ { "name": "@codexclaw/provider-bridge", - "version": "0.2.39", + "version": "0.2.40", "private": true, "type": "module", "description": "Detect-only opencodex (ocx) status probe at session start; graceful native path when absent.", diff --git a/plugins/codexclaw/components/recall/package.json b/plugins/codexclaw/components/recall/package.json index 1978765b..25f536e8 100644 --- a/plugins/codexclaw/components/recall/package.json +++ b/plugins/codexclaw/components/recall/package.json @@ -1,6 +1,6 @@ { "name": "@codexclaw/recall", - "version": "0.2.39", + "version": "0.2.40", "private": true, "type": "module", "description": "Read-only chat/memory recall search over the Codex session root (~/.codex): date-pruned rollout scan + thread/memory sqlite enrichment.", diff --git a/plugins/codexclaw/components/skill-search/package.json b/plugins/codexclaw/components/skill-search/package.json index 65726187..c483b427 100644 --- a/plugins/codexclaw/components/skill-search/package.json +++ b/plugins/codexclaw/components/skill-search/package.json @@ -1,6 +1,6 @@ { "name": "@codexclaw/skill-search", - "version": "0.2.39", + "version": "0.2.40", "private": true, "type": "module", "description": "Remote dormant-skill search over cli-jaw-skills / Hermes / ClawHub / gh code search. Zero-dep, TTL-cached, adapter-preamble output. No local vendoring.", diff --git a/plugins/codexclaw/components/subagent-config/package.json b/plugins/codexclaw/components/subagent-config/package.json index f4b74372..292d2a28 100644 --- a/plugins/codexclaw/components/subagent-config/package.json +++ b/plugins/codexclaw/components/subagent-config/package.json @@ -1,6 +1,6 @@ { "name": "@codexclaw/subagent-config", - "version": "0.2.39", + "version": "0.2.40", "private": true, "type": "module", "description": "Stores subagent model/prompt config; serves it to the GUI and an MCP tool.", diff --git a/plugins/codexclaw/gui/package.json b/plugins/codexclaw/gui/package.json index c7a6d32d..ea7b36c8 100644 --- a/plugins/codexclaw/gui/package.json +++ b/plugins/codexclaw/gui/package.json @@ -1,6 +1,6 @@ { "name": "@codexclaw/gui", - "version": "0.2.39", + "version": "0.2.40", "private": true, "type": "module", "description": "codexclaw local dashboard (Vite + React) \u2014 subagent config, prompts, provider link bar.", diff --git a/plugins/codexclaw/inventory.json b/plugins/codexclaw/inventory.json index a468f6e2..d0615024 100644 --- a/plugins/codexclaw/inventory.json +++ b/plugins/codexclaw/inventory.json @@ -2,8 +2,8 @@ "schemaVersion": 1, "plugin": { "name": "codexclaw", - "manifestVersion": "0.2.39+codex.20260924082502", - "packageVersion": "0.2.39" + "manifestVersion": "0.2.40+codex.20260929183231", + "packageVersion": "0.2.40" }, "skills": [ { @@ -326,55 +326,55 @@ { "folder": "bg-wake", "packageName": "@codexclaw/bg-wake", - "version": "0.2.39", + "version": "0.2.40", "hasTests": true }, { "folder": "config-guard", "packageName": "@codexclaw/config-guard", - "version": "0.2.39", + "version": "0.2.40", "hasTests": true }, { "folder": "cxc-ops", "packageName": "@codexclaw/cxc-ops", - "version": "0.2.39", + "version": "0.2.40", "hasTests": true }, { "folder": "messenger-bridge", "packageName": "@codexclaw/messenger-bridge", - "version": "0.2.39", + "version": "0.2.40", "hasTests": true }, { "folder": "pabcd-state", "packageName": "@codexclaw/pabcd-state", - "version": "0.2.39", + "version": "0.2.40", "hasTests": true }, { "folder": "provider-bridge", "packageName": "@codexclaw/provider-bridge", - "version": "0.2.39", + "version": "0.2.40", "hasTests": true }, { "folder": "recall", "packageName": "@codexclaw/recall", - "version": "0.2.39", + "version": "0.2.40", "hasTests": true }, { "folder": "skill-search", "packageName": "@codexclaw/skill-search", - "version": "0.2.39", + "version": "0.2.40", "hasTests": true }, { "folder": "subagent-config", "packageName": "@codexclaw/subagent-config", - "version": "0.2.39", + "version": "0.2.40", "hasTests": true } ]