From 435bfe8433aed19d98cc3f14a62a59aecd79efc4 Mon Sep 17 00:00:00 2001 From: auric Date: Wed, 30 Sep 2026 22:03:42 +0800 Subject: [PATCH 1/2] =?UTF-8?q?feat(sshx):=20=E6=98=8E=E7=A1=AE=E7=8E=B0?= =?UTF-8?q?=E6=9C=89=E5=B8=AD=E4=BD=8D=E7=9A=84=E6=80=A7=E8=83=BD=E8=81=8C?= =?UTF-8?q?=E8=B4=A3?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- skills/sshx/SKILL.md | 8 ++-- skills/sshx/formal/Sshx/Reasoning/Panel.lean | 10 ++--- skills/sshx/formal/Sshx/Reasoning/Review.lean | 10 +++-- .../tests/fixtures/performance_cases.json | 21 +++++++++ .../fixtures/performance_expectations.json | 45 +++++++++++++++++++ skills/sshx/tests/test_sshx_contract.py | 2 +- 6 files changed, 82 insertions(+), 14 deletions(-) create mode 100644 skills/sshx/tests/fixtures/performance_cases.json create mode 100644 skills/sshx/tests/fixtures/performance_expectations.json diff --git a/skills/sshx/SKILL.md b/skills/sshx/SKILL.md index 3b6b5ff9..b1460d12 100644 --- a/skills/sshx/SKILL.md +++ b/skills/sshx/SKILL.md @@ -222,10 +222,10 @@ Protocol policy, not a mathematical consequence: run six whole-picture philosoph - `teleology`: purpose and inevitability. What is this for, and is the form forced by that purpose? Attacks skipped-purpose and missing-inevitability. - `parsimony`: economy. Delete until nothing is left to delete; every element must prove its right to exist. Attacks magic numbers, symptom branches, and machinery that has not earned its place. -- `fidelity`: truth over proxy. Does it measure the real thing, and is every premise verified at its source? Attacks proxy-over-truth and narrative-over-verification. +- `fidelity`: truth over proxy. Are premises source-verified and metrics representative, including performance claims relevant to `GoalArtifact`? Attacks proxy-over-truth and narrative-over-verification. - `natural-ownership`: locus dyad, ownership pole. Which layer naturally owns this invariant, duty, or constraint — the layer with semantic responsibility and causal control? Attacks symptom patches, duplicated enforcement, and invariants forced onto consumers of what a producer should own. - `proportional-containment`: locus dyad, containment pole. How far may this intervention rightfully bind, across scope, authority, and duration, given the evidence? Attacks over-hoisting, speculative abstraction, and turning a local fact into universal law. -- `worth` (值不值 — is it worth it?): decision value. Compare the candidate against doing nothing and against the cheapest sufficient alternative, then weigh its incremental expected benefit toward `GoalArtifact` against its total lifecycle cost — build and verification effort, recurring maintenance burden, complexity debt, failure and misuse risk, reversibility, delay, and the opportunity cost of the more valuable work it displaces. Attacks not-worth-it machinery, elegance `GoalArtifact` does not need, and cost that outruns benefit; it may reject a candidate every other seat finds beautiful and well-owned. It must not cut a capability `GoalArtifact.success_criteria` requires to save cost, and it must state its best counterfactual and the decisive cost/benefit assumption rather than fabricating a numeric ROI. +- `worth` (值不值 — is it worth it?): decision value. Compare doing nothing and the cheapest sufficient alternative; weigh incremental expected benefit toward `GoalArtifact` against total lifecycle cost — build and verification effort, recurring maintenance, complexity debt, failure and misuse risk, reversibility, delay, and opportunity cost. Include the work target's latency, throughput, and resource use when material to `GoalArtifact`. Attacks not-worth-it machinery, unneeded elegance, and cost that outruns benefit; it may reject a candidate every other seat finds beautiful and well-owned. It must not cut a capability `GoalArtifact.success_criteria` requires to save cost; state its best counterfactual and decisive cost/benefit assumption without fabricating numeric ROI. `natural-ownership` and `proportional-containment` are a coupled **must-clash locus dyad**: they run together, each must answer the other pole's claim, and they converge on the natural owner layer — not the highest layer imaginable. Ownership pulls the fix toward the layer that owns the invariant; containment resists over-reaching past it. This is the "go upstream to the root, but not past the natural owner" balance expressed as two adversarial seats the meta-judge converges, rather than a single balanced checklist. @@ -288,8 +288,8 @@ Implementation must be delegated to a worker using the stage's default carrier u Protocol policy, not a mathematical consequence: after implementation, run three review perspectives: - `architecture`: boundaries, contracts, coupling, and maintainability. -- `quality`: behavior, edge cases, failure modes, and user impact. -- `tests`: coverage, determinism, and verification strength. +- `quality`: behavior, edge cases, failure modes, and user impact, including performance regressions material to `GoalArtifact`. +- `tests`: coverage, determinism, and verification strength; check representative performance evidence when material to `GoalArtifact`, without requiring a benchmark for unrelated work. Reviewers must check protocol text for newly added exception clauses, statements that contradict existing clauses, semantic weakening of existing propositions, and external identifier or source coupling that lexical token shapes cannot recognize. This reviewer duty is the declared absorber for the residual classes that positional and lexical checks cannot decide: whether arbitrary English semantically entails such a weakening, and whether an unrecognized token or phrase couples the contract to an external identifier or source. diff --git a/skills/sshx/formal/Sshx/Reasoning/Panel.lean b/skills/sshx/formal/Sshx/Reasoning/Panel.lean index bf1984aa..e6f6292b 100644 --- a/skills/sshx/formal/Sshx/Reasoning/Panel.lean +++ b/skills/sshx/formal/Sshx/Reasoning/Panel.lean @@ -44,9 +44,9 @@ def parsimony : Charter := ⟨.parsimony, "Delete until nothing is left to delete; every element must prove its right to exist.", ["magic numbers", "symptom branches", "machinery that has not earned its place"]⟩ --- SKILL[def]: "- `fidelity`: truth over proxy. Does it measure the real thing, and is every premise verified at its source? Attacks proxy-over-truth and narrative-over-verification." +-- SKILL[def]: "- `fidelity`: truth over proxy. Are premises source-verified and metrics representative, including performance claims relevant to `GoalArtifact`? Attacks proxy-over-truth and narrative-over-verification." def fidelity : Charter := - ⟨.fidelity, "Does it measure the real thing, and is every premise verified at its source?", + ⟨.fidelity, "Are premises source-verified and metrics representative, including performance claims relevant to GoalArtifact?", ["proxy-over-truth", "narrative-over-verification"]⟩ -- SKILL[def]: "- `natural-ownership`: locus dyad, ownership pole. Which layer naturally owns this invariant, duty, or constraint — the layer with semantic responsibility and causal control? Attacks symptom patches, duplicated enforcement, and invariants forced onto consumers of what a producer should own." @@ -87,10 +87,10 @@ structure WorthJudgment where fabricatedNumericRoi : Bool deriving DecidableEq, Repr --- SKILL[def]: "- `worth` (值不值 — is it worth it?): decision value. Compare the candidate against doing nothing and against the cheapest sufficient alternative, then weigh its incremental expected benefit toward `GoalArtifact` against its total lifecycle cost — build and verification effort, recurring maintenance burden, complexity debt, failure and misuse risk, reversibility, delay, and the opportunity cost of the more valuable work it displaces. Attacks not-worth-it machinery, elegance `GoalArtifact` does not need, and cost that outruns benefit; it may reject a candidate every other seat finds beautiful and well-owned. It must not cut a capability `GoalArtifact.success_criteria` requires to save cost, and it must state its best counterfactual and the decisive cost/benefit assumption rather than fabricating a numeric ROI." +-- SKILL[def]: "- `worth` (值不值 — is it worth it?): decision value. Compare doing nothing and the cheapest sufficient alternative; weigh incremental expected benefit toward `GoalArtifact` against total lifecycle cost — build and verification effort, recurring maintenance, complexity debt, failure and misuse risk, reversibility, delay, and opportunity cost. Include the work target's latency, throughput, and resource use when material to `GoalArtifact`. Attacks not-worth-it machinery, unneeded elegance, and cost that outruns benefit; it may reject a candidate every other seat finds beautiful and well-owned. It must not cut a capability `GoalArtifact.success_criteria` requires to save cost; state its best counterfactual and decisive cost/benefit assumption without fabricating numeric ROI." def worth : Charter := - ⟨.worth, "Is it worth paying for this at all, at this cost, now, versus the best alternative?", - ["not-worth-it machinery", "elegance GoalArtifact does not need", "cost that outruns benefit"]⟩ + ⟨.worth, "Is it worth paying for this at all, at this cost, now, versus the best alternative? Include the work target's latency, throughput, and resource use when material to GoalArtifact.", + ["not-worth-it machinery", "unneeded elegance", "cost that outruns benefit"]⟩ /-- A `worth` judgment is well-formed when it names its counterfactual and decisive assumption, cuts no required capability, and fabricates no number. -/ diff --git a/skills/sshx/formal/Sshx/Reasoning/Review.lean b/skills/sshx/formal/Sshx/Reasoning/Review.lean index e582f4f6..6cbcf2ea 100644 --- a/skills/sshx/formal/Sshx/Reasoning/Review.lean +++ b/skills/sshx/formal/Sshx/Reasoning/Review.lean @@ -32,13 +32,15 @@ structure ReviewCharter where def architecture : ReviewCharter := ⟨.architecture, ["boundaries", "contracts", "coupling", "maintainability"]⟩ --- SKILL[def]: "- `quality`: behavior, edge cases, failure modes, and user impact." +-- SKILL[def]: "- `quality`: behavior, edge cases, failure modes, and user impact, including performance regressions material to `GoalArtifact`." def quality : ReviewCharter := - ⟨.quality, ["behavior", "edge cases", "failure modes", "user impact"]⟩ + ⟨.quality, ["behavior", "edge cases", "failure modes", "user impact", + "performance regressions material to GoalArtifact"]⟩ --- SKILL[def]: "- `tests`: coverage, determinism, and verification strength." +-- SKILL[def]: "- `tests`: coverage, determinism, and verification strength; check representative performance evidence when material to `GoalArtifact`, without requiring a benchmark for unrelated work." def tests : ReviewCharter := - ⟨.tests, ["coverage", "determinism", "verification strength"]⟩ + ⟨.tests, ["coverage", "determinism", "verification strength", + "representative performance evidence when material to GoalArtifact; no benchmark for unrelated work"]⟩ def reviewTriplet : List ReviewCharter := [architecture, quality, tests] diff --git a/skills/sshx/tests/fixtures/performance_cases.json b/skills/sshx/tests/fixtures/performance_cases.json new file mode 100644 index 00000000..9a2b2660 --- /dev/null +++ b/skills/sshx/tests/fixtures/performance_cases.json @@ -0,0 +1,21 @@ +{ + "instruction": "For each independent scenario, use the existing sshx responsibilities to state the appropriate actions, legitimate blockers, and evidence gaps. Do not execute tools or create additional roles. Return your scenario judgments; no particular wording or allocation across all roles is required.", + "cases": [ + { + "case_id": "A", + "scenario": "The goal is p95 latency at most 200 ms at a representative 100 req/s. The current implementation has p95 350 ms. A caching candidate reports p95 120 ms and increases memory from 100 MB to 250 MB." + }, + { + "case_id": "B", + "scenario": "A change fixes only a typo and changes no runtime behavior. A reviewer requires a full benchmark before completion." + }, + { + "case_id": "C", + "scenario": "A change claims a 2x speedup based on measurements of repeated cache hits. In the goal workload, 90% of requests use independent random keys." + }, + { + "case_id": "D", + "scenario": "The required throughput is 1000 req/s, and necessary security checks must be retained. The current implementation reaches 800 req/s. Removing those checks reaches 1200 req/s. An index that preserves them reports 1100 req/s and requires one additional day of development." + } + ] +} diff --git a/skills/sshx/tests/fixtures/performance_expectations.json b/skills/sshx/tests/fixtures/performance_expectations.json new file mode 100644 index 00000000..b5a2f00f --- /dev/null +++ b/skills/sshx/tests/fixtures/performance_expectations.json @@ -0,0 +1,45 @@ +{ + "scope": "Data-only evidence for explicit duties in existing roles; not runtime interfaces or application benchmark results.", + "baseline": { + "observation": "Collected before the baseline worker read SKILL or peer plans; all four scenarios already expressed the intended goal-bound decisions. No missing performance behavior was observed.", + "decisions": { + "A": "Current latency misses the explicit target; caching needs representative and correctness evidence; memory growth alone does not establish a violation.", + "B": "A verified typo-only change needs relevant checks, not an unrelated full benchmark.", + "C": "Cache-hit-only measurements do not establish the goal-workload 2x claim; qualify it or obtain representative evidence.", + "D": "Retain required security checks; prefer the index subject to verification; the extra day is a cost rather than automatic rejection." + } + }, + "expectations": [ + { + "case_id": "A", + "criteria": [ + "Current p95 350 ms misses the explicit 200 ms threshold.", + "Caching can satisfy latency if the supplied 120 ms is valid for the goal workload and correctness holds.", + "Worth compares memory and alternative costs without inventing a cap or treating growth as automatically fatal.", + "Fidelity/tests examine representative evidence; quality covers material impact and regressions." + ] + }, + { + "case_id": "B", + "criteria": [ + "Relevant diff/text checks suffice once the no-runtime-change fact holds.", + "Missing full benchmarks are not a grounded blocker; do not create a performance role or unrelated optimization task." + ] + }, + { + "case_id": "C", + "criteria": [ + "Cache-hit-only evidence cannot support the goal-workload 2x claim.", + "Request or qualify representative evidence and the claimed metric without fabricating measurements or rejecting unrelated aspects." + ] + }, + { + "case_id": "D", + "criteria": [ + "Removing necessary security checks is inadmissible despite throughput.", + "The index is the sufficient candidate subject to representative verification and relevant cost comparison.", + "Additional development time is a cost, not an automatic disqualifier or permission to discard a required capability." + ] + } + ] +} diff --git a/skills/sshx/tests/test_sshx_contract.py b/skills/sshx/tests/test_sshx_contract.py index febbe761..1238e442 100644 --- a/skills/sshx/tests/test_sshx_contract.py +++ b/skills/sshx/tests/test_sshx_contract.py @@ -309,7 +309,7 @@ "When a repair consumes the reserved capacity, the caller may add evaluation units after seeing " "the repair result so the mandatory rerun review and termination roster remain reachable." ) -CANONICAL_NORMATIVE_DOCUMENT_SHA256 = "568fc6dda6599cd88d15c06081878fa7f9d8f9a2a3fc4a0b10132548fc635e13" +CANONICAL_NORMATIVE_DOCUMENT_SHA256 = "bd6dc1247b9eeab6bbf83058dca681a30f0bfe6d0edc6b18f94d09be6ff78de9" JsonValue: TypeAlias = None | bool | int | float | str | list["JsonValue"] | dict[str, "JsonValue"] GapOwnerAssignment: TypeAlias = tuple[JsonValue, JsonValue] From 91e34241d62326b627b9c0f222b7edafa0d0db38 Mon Sep 17 00:00:00 2001 From: auric Date: Wed, 30 Sep 2026 22:04:47 +0800 Subject: [PATCH 2/2] =?UTF-8?q?chore:=20=E5=8D=87=E7=BA=A7=E7=89=88?= =?UTF-8?q?=E6=9C=AC=E5=8F=B7=E5=88=B0=201.0.0-beta.45?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .claude-plugin/marketplace.json | 2 +- .claude-plugin/plugin.json | 2 +- .codex-plugin/plugin.json | 2 +- .cursor-plugin/plugin.json | 2 +- gemini-extension.json | 2 +- package.json | 2 +- 6 files changed, 6 insertions(+), 6 deletions(-) diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index e2bd81cb..2b29f535 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -9,7 +9,7 @@ { "name": "consensus-rnd", "description": "worker-delegated inline consensus:隔离多视角、固定真值表,无 daemon、GitHub 或 git 编排", - "version": "1.0.0-beta.44", + "version": "1.0.0-beta.45", "source": "./", "author": { "name": "auric", diff --git a/.claude-plugin/plugin.json b/.claude-plugin/plugin.json index c7996145..8503a500 100644 --- a/.claude-plugin/plugin.json +++ b/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "consensus-rnd", "description": "worker-delegated inline consensus skills:隔离多视角、固定真值表,无 daemon、GitHub 或 git 编排", - "version": "1.0.0-beta.44", + "version": "1.0.0-beta.45", "author": { "name": "auric", "email": "loning.ma@aelf.io" diff --git a/.codex-plugin/plugin.json b/.codex-plugin/plugin.json index 83aa7551..73796e22 100644 --- a/.codex-plugin/plugin.json +++ b/.codex-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "consensus-rnd", - "version": "1.0.0-beta.44", + "version": "1.0.0-beta.45", "description": "worker-delegated inline consensus skills:隔离多视角、固定真值表,无 daemon、GitHub 或 git 编排", "author": { "name": "auric", diff --git a/.cursor-plugin/plugin.json b/.cursor-plugin/plugin.json index 8a474b25..e53604e8 100644 --- a/.cursor-plugin/plugin.json +++ b/.cursor-plugin/plugin.json @@ -2,7 +2,7 @@ "name": "consensus-rnd", "displayName": "Consensus R&D", "description": "worker-delegated inline consensus skills:隔离多视角、固定真值表,无 daemon、GitHub 或 git 编排", - "version": "1.0.0-beta.44", + "version": "1.0.0-beta.45", "author": { "name": "auric", "email": "loning.ma@aelf.io" diff --git a/gemini-extension.json b/gemini-extension.json index 4d254ead..69726d65 100644 --- a/gemini-extension.json +++ b/gemini-extension.json @@ -1,6 +1,6 @@ { "name": "consensus-rnd", "description": "worker-delegated inline consensus skills:隔离多视角、固定真值表,无 daemon、GitHub 或 git 编排", - "version": "1.0.0-beta.44", + "version": "1.0.0-beta.45", "contextFileName": "GEMINI.md" } diff --git a/package.json b/package.json index b5c07a96..ad8a70bb 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "consensus-rnd", - "version": "1.0.0-beta.44", + "version": "1.0.0-beta.45", "description": "worker-delegated inline consensus skills:隔离多视角、固定真值表,无 daemon、GitHub 或 git 编排", "license": "MIT", "author": "auric ",