Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 13 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -56,6 +56,19 @@ breaking changes may land in a minor release.
child), except a share-root shape older PowerShell would corrupt.
- Escalate an environment fault at the review-budget rescue gate instead of
deferring the story as unconverged (DW-523).
- Reach only control-session windows carrying this project's tag from `a`/`x`,
so a reused window id or a forged `ctl-window` record can no longer steer
them onto a neighbour's window; the record now only breaks ties among tagged
windows, and a window whose tag write failed is left alone until relaunched
(#750). When a window under the run's name is left alone because its tag reads
empty (never written, or unreadable — psmux's option probe can fail), `x`
warns that the control window was not closed instead of reporting a clean
stop, and `a` and `bmad-loop attach` say why they cannot reach it. A control
listing that fails outright is reported as such instead of reading as "no
window" (by `x`, `a`, `bmad-loop attach` and the window prune) — `a` and
`bmad-loop attach` then still reach the run's agent session — `x` checks
that the window it killed is gone, and a TUI run or sweep launch warns when
its new window cannot be confirmed as this project's.
- Explain that unpinned result-artifact scans search only the configured artifact
directories themselves, so a nested story spec no longer produces an opaque
`no-artifact` breadcrumb (#780).
Expand Down
7 changes: 6 additions & 1 deletion docs/tui-guide.md
Original file line number Diff line number Diff line change
Expand Up @@ -88,7 +88,12 @@ The TUI never runs an engine in-process. The two halves:
swept with `c` (see [Cleaning up sessions](#cleaning-up-sessions-c)). Each
launch over an existing run records the id of the window it minted in the
run dir (`ctl-window`), so attach/stop follow the run's live window even
while an older same-run-id window is still parked (#482).
while an older same-run-id window is still parked (#482). Attach/stop only
ever reach a window carrying this project's tag: the record breaks ties
among tagged windows but never vouches for an untagged one, so a window
whose tag write failed is left alone until a relaunch tags one (#750). The
same holds when the tag cannot be read at all, and in both cases `x` warns
that the control window was not closed and `a` says why it cannot reach it.
- **Observer** — the dashboard reads only the artifacts the engine writes
atomically into `.bmad-loop/runs/<run-id>/`: `state.json`, `journal.jsonl`,
`logs/<task-id>.log`, `ATTENTION`, `engine.pid`. It polls the selected run
Expand Down
22 changes: 20 additions & 2 deletions src/bmad_loop/cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -5452,9 +5452,27 @@ def cmd_attach(args: argparse.Namespace) -> int:
if run_dir is None:
print("no runs found", file=sys.stderr)
return 1
plan = launch.attach_plan(project, run_dir.name)
# A ctl listing that could not be read is said, then the attach carries on
# to the agent session rather than failing on a window it could not check.
ctl_faults: list[str] = []

def ctl_fault(msg: str) -> None:
ctl_faults.append(msg)
print(f"warning: could not check the run's control window: {msg}", file=sys.stderr)

plan, unproven = launch.attach_plan(project, run_dir.name, on_fault=ctl_fault)
if unproven:
# A window under this run's name was refused for an empty tag (#750):
# say so before attaching elsewhere — or reporting nothing — so the
# refusal is not mistaken for an absence. Before the attach, which takes
# over the terminal.
notice = launch.unproven_ctl_window_notice(project, run_dir.name, unproven)
print(f"warning: {notice}", file=sys.stderr)
if plan is None:
print(f"nothing to attach for run {run_dir.name}", file=sys.stderr)
if not ctl_faults:
# Not after a ctl fault: "nothing to attach" would claim the
# unchecked ctl window absent (#750). The warning above stands.
print(f"nothing to attach for run {run_dir.name}", file=sys.stderr)
return 1
argv, return_window = plan
# Record where to send the client once the sweep finishes this cycle's
Expand Down
98 changes: 88 additions & 10 deletions src/bmad_loop/tui/app.py
Original file line number Diff line number Diff line change
Expand Up @@ -338,7 +338,7 @@ def _start_run_result(self, result: dict | None) -> None:
def go() -> None:
run_id = runs.new_run_id()
try:
launch.start_run_detached(
win_id = launch.start_run_detached(
self.project,
run_id,
spec=spec_folder or None,
Expand All @@ -349,13 +349,29 @@ def go() -> None:
except launch.LaunchError as e:
self.notify(str(e), severity="error")
return
if not win_id:
self._warn_unreachable_launch("run", run_id)
self.notify(
f"run {run_id} launched (control session {launch.ctl_session(self.project)})"
)
self._dashboard.expect_run(run_id)

self._guarded(go)

def _warn_unreachable_launch(self, kind: str, run_id: str) -> None:
"""The launch is running, but its ctl window could not be confirmed
reachable by the lookup `a`/`x` use — its best-effort tag write did not
land (#750), its id was not captured, or the listing could not be read.
Say so beside the success toast instead of letting it imply the
targeting is sound."""
self.notify(
f"{kind} {run_id} launched, but its control window could not be confirmed "
"as this project's — attach/stop may not reach it; check "
f"{launch.ctl_session(self.project)} and close it by hand when done",
severity="warning",
timeout=15,
)

def action_start_sweep(self) -> None:
if self._mux_missing():
return
Expand All @@ -374,7 +390,7 @@ def _start_sweep_result(self, result: dict | None) -> None:
def go() -> None:
run_id = runs.new_run_id()
try:
launch.start_sweep_detached(
win_id = launch.start_sweep_detached(
self.project,
run_id,
no_prompt=result["no_prompt"],
Expand All @@ -384,6 +400,8 @@ def go() -> None:
except launch.LaunchError as e:
self.notify(str(e), severity="error")
return
if not win_id:
self._warn_unreachable_launch("sweep", run_id)
self.notify(
f"sweep {run_id} launched (control session {launch.ctl_session(self.project)})"
)
Expand Down Expand Up @@ -549,22 +567,52 @@ def action_attach(self) -> None:
self.notify("no run selected", severity="warning")
return
session = runs.session_name(run_id)
win_id = launch.ctl_window_id(self.project, run_id)
# A ctl listing that failed raises rather than reading as "no window"
# (#750). Say so, but do not abort: the agent session may still be
# reachable, so carry on exactly as if there were no ctl window.
ctl_fault: str | None = None
try:
win_id, unproven = launch.ctl_window_lookup(self.project, run_id)
except MultiplexerError as e:
win_id, unproven, ctl_fault = None, 0, str(e)
self.notify(
f"could not check the run's control window: {e}",
severity="warning",
timeout=15,
)
ok, agent_live = self._mux_guarded(lambda: launch.agent_session_exists(session))
if not ok:
return
# A sweep blocked on a decision prompt has no agent session — the
# human answers in the orchestrator's ctl window. Otherwise prefer the
# live agent session, falling back to the ctl window between sessions.
if win_id is not None and (self._dashboard.decision_pending is not None or not agent_live):
wants_ctl = self._dashboard.decision_pending is not None or not agent_live
if unproven:
# A window under this run's name was refused because its tag could
# not be read as ours (#750) — possibly the live orchestrator, even
# beside a tagged window answered here. Say so whatever is attached.
lead = (
"cannot attach to the run window"
if win_id is None and wants_ctl
else "attaching without a window it could not prove"
)
self.notify(
f"{lead}: {launch.unproven_ctl_window_notice(self.project, run_id, unproven)}",
severity="warning",
timeout=15,
)
if win_id is not None and wants_ctl:
launch.select_ctl_window_id(win_id)
self._attach_to_target(launch.ctl_target(self.project), return_window=win_id)
return
elif agent_live:
if agent_live:
target = runs.session_target(run_id)
else:
# Textual captures stderr, so agent_session_exists' warning about a
# same-named session of another project's never reaches the screen.
# Checked BEFORE the ctl early returns below: the ctl warning says why
# the window is out of reach, and only this says why the agent
# session is too.
ok, refusal = self._mux_guarded(lambda: runs.foreign_session_refusal(session))
if not ok:
return
Expand All @@ -574,6 +622,10 @@ def action_attach(self) -> None:
f"not attaching: {refusal}", severity="warning", timeout=10, markup=False
)
return
if unproven:
return # the warning above already said why
if ctl_fault is not None:
return # "no ctl window" would be a claim the fault toast unsays
self.notify(
f"nothing to attach: no live agent session ({session}) and no "
f"{launch.ctl_session(self.project)} window for this run (runs started outside "
Expand Down Expand Up @@ -701,8 +753,9 @@ def _launch_resolve(self, run_id: str) -> None:
# is lost is the record *later* verbs read, so `a`/`x` after this
# window is minted may answer an older one (#482's symptom).
self.notify(
"resolve launched but its window id was not recorded — "
"later attach/stop may target an older window for this run",
"resolve launched but its window id was not recorded or its tag did "
"not land — later attach/stop may miss it or target an older window "
"for this run",
severity="warning",
)
launch.select_ctl_window_id(win_id)
Expand Down Expand Up @@ -972,8 +1025,9 @@ def _do_resume(self, run_id: str) -> None:
# uncaptured id and the unwritten record through this one signal
# because they leave the operator in the same place.
self.notify(
"resume launched but its window id was not recorded — "
"attach/stop may target an older window for this run",
"resume launched but its window id was not recorded or its tag did "
"not land — attach/stop may miss it or target an older window for "
"this run",
severity="warning",
)
self.notify(
Expand Down Expand Up @@ -1567,7 +1621,6 @@ def done(ok: bool | None) -> None:
def _stop_run_worker(self, run_id: str, run_dir: Path) -> None:
try:
runs.stop_run(run_dir)
launch.kill_ctl_window(self.project, run_id)
except (OSError, StopRunError, ProcessHostError) as e:
self.call_from_thread(self.notify, f"stop failed: {e}", severity="error")
return
Expand All @@ -1580,6 +1633,31 @@ def _stop_run_worker(self, run_id: str, run_dir: Path) -> None:
severity="warning",
markup=False,
)
try:
left = launch.kill_ctl_window(self.project, run_id)
except (OSError, MultiplexerError, UnicodeError) as e:
# The engine is stopped; only the ctl-window half failed — its listing
# could not be read, or the window survived the kill (#750). A plain
# "stopped" would claim that window is gone.
self.call_from_thread(
self.notify,
f"run {run_id} stopped, but its control window may still be running: {e}",
severity="warning",
timeout=15,
)
return
if left:
# The engine stopped, but a ctl window under its name may still be
# running — even when another one was closed: a plain "stopped"
# would hide that (#750).
self.call_from_thread(
self.notify,
f"run {run_id} stopped, but a control window under its name was not closed: "
f"{launch.unproven_ctl_window_notice(self.project, run_id, left)}",
severity="warning",
timeout=15,
)
return
self.call_from_thread(self.notify, f"run {run_id} stopped")

def action_graceful_stop_run(self) -> None:
Expand Down
Loading
Loading