diff --git a/.github/workflows/integration-test.yml b/.github/workflows/integration-test.yml index 5df61fc4c..8ceb9dacc 100644 --- a/.github/workflows/integration-test.yml +++ b/.github/workflows/integration-test.yml @@ -954,11 +954,14 @@ jobs: # pinned claude package's own installer, which fetches the native # binary from the vendor's release endpoint — the same vendor trust # executing the agent itself carries. Keep the pins in lockstep with - # run-predict, run-evaluate, and run-backtest. + # the workflows that install the same CLIs: run-predict and + # run-evaluate (gemini), run-backtest (all three), and run-analytics's + # labeler (claude). Production claude cells run the CLI their action + # bundles instead, which the engine-actions-smoke leg exercises. run: | set -euo pipefail case "$ENGINE" in - claude-code) package="@anthropic-ai/claude-code@2.1.259" binary=claude ;; + claude-code) package="@anthropic-ai/claude-code@2.1.280" binary=claude ;; codex) package="@openai/codex@0.144.1" binary=codex ;; gemini) package="@google/gemini-cli@0.49.0" binary=gemini ;; esac @@ -2868,7 +2871,7 @@ jobs: # expressible for `npm i -g`. run: | set -euo pipefail - npm install --silent --no-audit --ignore-scripts --global @anthropic-ai/claude-code@2.1.259 + npm install --silent --no-audit --ignore-scripts --global @anthropic-ai/claude-code@2.1.280 node "$(npm root -g)/@anthropic-ai/claude-code/install.cjs" claude --version path=$(command -v claude) diff --git a/.github/workflows/run-analytics.yml b/.github/workflows/run-analytics.yml index 24d3ec2a2..fb297ba82 100644 --- a/.github/workflows/run-analytics.yml +++ b/.github/workflows/run-analytics.yml @@ -139,6 +139,7 @@ on: - claude-haiku-4-5-20251001 - claude-sonnet-4-6 - claude-fable-5 + - claude-opus-5-5 default: claude-haiku-4-5-20251001 # corpus-stats inputs (ignored by every other mode). All are optional; an # empty value means "no filter" / "no grouping". @@ -1233,9 +1234,11 @@ jobs: # instead — the same npm pattern run-backtest.yml and integration-test.yml # prove on these runners, kept in version step with both — and hand the # action the executable, which skips its own install path entirely. The - # action bundles a slightly newer CLI for the cells it installs itself; - # if the labeler ever misbehaves under this pin, matching the bundled - # version is the first lever. + # pin runs ahead of the CLI the action bundles, because it is the floor + # the newest `label_model` choice needs (an older CLI refuses the model + # id). If the labeler ever misbehaves under this pin, the first lever is + # an action release whose bundled agent SDK matches it — not a rollback, + # which would drop that choice. - name: Install the pinned Claude Code CLI for the labeler id: claude-cli timeout-minutes: 5 @@ -1248,7 +1251,7 @@ jobs: # executing the agent itself carries. run: | set -euo pipefail - npm install --silent --no-audit --ignore-scripts --global @anthropic-ai/claude-code@2.1.259 + npm install --silent --no-audit --ignore-scripts --global @anthropic-ai/claude-code@2.1.280 # Claude's native binary is placed by its postinstall, which # --ignore-scripts suppresses; run the pinned package's own # installer explicitly. diff --git a/.github/workflows/run-backtest.yml b/.github/workflows/run-backtest.yml index a136f7ef2..e8c4c4e19 100644 --- a/.github/workflows/run-backtest.yml +++ b/.github/workflows/run-backtest.yml @@ -457,7 +457,7 @@ jobs: run: | set -euo pipefail npm install --silent --no-audit --ignore-scripts --global \ - @anthropic-ai/claude-code@2.1.259 @openai/codex@0.144.1 \ + @anthropic-ai/claude-code@2.1.280 @openai/codex@0.144.1 \ @google/gemini-cli@0.49.0 # Claude's native binary is placed by its postinstall, which # --ignore-scripts suppresses; run the pinned package's own diff --git a/docs/pipeline.md b/docs/pipeline.md index b8d8de25e..8b6ed81b7 100644 --- a/docs/pipeline.md +++ b/docs/pipeline.md @@ -471,6 +471,7 @@ queues behind the production run of the same mode. The modes: the measured block still reaches the step summary, and the job fails. The `label_model` dispatch input picks the labeler's model; a ceiling-sized run overrides the default for `claude-fable-5`, the one tier measured to finish. + `claude-opus-5-5` is offered too, with no measured pace or cost yet. See [qp-topic.md](qp-topic.md). ## `integration-test` — the infrastructure preflight