From 2e3f323ef61f3f9f8bcf6ef46c1c62319fbdf259 Mon Sep 17 00:00:00 2001 From: XuKun Cai Date: Tue, 29 Sep 2026 21:41:24 +0800 Subject: [PATCH] release: prepare SkillBench 1.1.0 and guarded main publication --- .github/workflows/publish.yml | 23 ++++++------ CHANGELOG.md | 10 ++++++ README.md | 16 ++++----- README_EN.md | 16 ++++----- RELEASING.md | 4 +-- docs/demo.svg | 2 +- package.json | 2 +- src/core/token-analyzer.ts | 54 +++++++++++++++-------------- src/version.ts | 2 +- tests/unit/package-metadata.test.ts | 6 ++-- tests/unit/public-api.test.ts | 2 +- tests/unit/publish-workflow.test.ts | 5 +-- tests/unit/token-analyzer.test.ts | 16 +++++++++ 13 files changed, 95 insertions(+), 63 deletions(-) diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml index b5e955c..e8a560e 100644 --- a/.github/workflows/publish.yml +++ b/.github/workflows/publish.yml @@ -24,7 +24,7 @@ permissions: contents: read concurrency: - group: ${{ github.workflow }}-${{ github.ref }} + group: ${{ github.workflow }}-${{ github.repository }} cancel-in-progress: false jobs: @@ -48,7 +48,7 @@ jobs: - run: pnpm install --frozen-lockfile - run: pnpm release:check - bootstrap-1-0: + publish-main: if: github.event_name == 'push' && github.ref == 'refs/heads/main' runs-on: ubuntu-latest permissions: @@ -68,7 +68,7 @@ jobs: - run: pnpm install --frozen-lockfile - run: pnpm release:check - id: release - name: Resolve bootstrap release state + name: Resolve release state shell: bash run: | set -euo pipefail @@ -77,11 +77,7 @@ jobs: expected_repo='git+https://github.com/EpochTX/skillbench.git' echo "version=$version" >> "$GITHUB_OUTPUT" echo "package=$package" >> "$GITHUB_OUTPUT" - if [[ "$version" != '1.0.0' ]]; then - echo 'eligible=false' >> "$GITHUB_OUTPUT" - echo 'already_published=false' >> "$GITHUB_OUTPUT" - exit 0 - fi + [[ "$version" =~ ^[0-9]+\.[0-9]+\.[0-9]+$ ]] echo 'eligible=true' >> "$GITHUB_OUTPUT" if npm view "$package@$version" version --json >/dev/null 2>&1; then existing_repo="$(npm view "$package@$version" repository.url --json)" @@ -90,10 +86,15 @@ jobs: exit 1 fi echo 'already_published=true' >> "$GITHUB_OUTPUT" + if gh api "repos/$GITHUB_REPOSITORY/git/ref/tags/v$version" >/dev/null 2>&1; then + echo 'eligible=false' >> "$GITHUB_OUTPUT" + fi else echo 'already_published=false' >> "$GITHUB_OUTPUT" fi - - name: Publish 1.0.0 to npm + env: + GH_TOKEN: ${{ github.token }} + - name: Publish package to npm if: steps.release.outputs.eligible == 'true' && steps.release.outputs.already_published != 'true' run: npm publish --access public --provenance env: @@ -131,7 +132,7 @@ jobs: printf '# Agent Instructions\n\nKeep changes scoped, deterministic, and reviewable.\n' > SKILL.md ./node_modules/.bin/skillbench scan SKILL.md --format json --output report.json node --input-type=module -e "import fs from 'node:fs'; const report=JSON.parse(fs.readFileSync('report.json','utf8')); if (report.tool?.version !== '$version') process.exit(1);" - - name: Create v1.0.0 tag and GitHub release + - name: Create version tag and GitHub release if: steps.release.outputs.eligible == 'true' shell: bash env: @@ -140,7 +141,7 @@ jobs: set -euo pipefail version='${{ steps.release.outputs.version }}' tag="v$version" - release_sha='365533eb2feb60467336ce1faca2df96a4ad1d78' + release_sha="$GITHUB_SHA" existing_sha='' if ref_json="$(gh api "repos/$GITHUB_REPOSITORY/git/ref/tags/$tag" 2>/dev/null)"; then existing_sha="$(jq -r '.object.sha' <<<"$ref_json")" diff --git a/CHANGELOG.md b/CHANGELOG.md index 61768a8..9d8af4b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,16 @@ All notable changes to SkillBench will be documented in this file. The project f ## Unreleased +## 1.1.0 — 2026-09-29 + +### Improved + +- Exact repeated paragraphs are detected throughout long instruction files, including beyond the 500-paragraph fuzzy-comparison budget. Near-duplicate checks remain bounded. +- Repository scans limit concurrent instruction-file reads and count per-file findings in one pass. The 120-file guard also checks that finding totals agree. +- CI, package-integrity checks, and the labeled rule benchmark continue to cover Node.js 20 and 22, Windows, and macOS. +- Scheduled Dependabot version-update pull requests are disabled; dependency upgrades now follow manual release review. +- The publish workflow can release a validated version bump from `main`, verify the npm registry and installed CLI, then create a tag and GitHub Release pointing at that exact commit. Tag-triggered runs remain idempotent. + ## 1.0.0 — 2026-08-12 ### Added diff --git a/README.md b/README.md index 4435b79..a317086 100644 --- a/README.md +++ b/README.md @@ -14,14 +14,14 @@ CI MIT license Node 20 or newer - SkillBench version 1.0.0 + SkillBench version 1.1.0

![SkillBench terminal demo](docs/demo.svg) SkillBench 1.0 将 Agent Skill、`AGENTS.md`、`CLAUDE.md`、Cursor Rules 等指令文件纳入可重复的工程质量体系:静态检查、安全规则、Token 效率、跨 Agent 兼容性、回归对比、SARIF、CI 门禁、可审计安全修复和人工标签规则 Benchmark。 -> **发布完整性保证:** `v1.0.0` 只会在 `skillbench-ai@1.0.0` 已成功发布并从 npm Registry 验证后创建。如果你正在查看尚未打 tag 的 release candidate,请使用下方“从源码运行”方式。 +> **发布完整性保证:** `v1.1.0` 只会在 `skillbench-ai@1.1.0` 已成功发布并从 npm Registry 验证后创建。如果你正在查看尚未打 tag 的 release candidate,请使用下方“从源码运行”方式。 ## 快速开始 @@ -30,13 +30,13 @@ SkillBench 1.0 将 Agent Skill、`AGENTS.md`、`CLAUDE.md`、Cursor Rules 等指 无需全局安装: ```bash -npx --yes skillbench-ai@1.0.0 scan SKILL.md +npx --yes skillbench-ai@1.1.0 scan SKILL.md ``` 或全局安装 CLI: ```bash -npm install --global skillbench-ai@1.0.0 +npm install --global skillbench-ai@1.1.0 skillbench scan SKILL.md ``` @@ -251,12 +251,12 @@ ignore: ## JSON / SARIF / API 稳定性 -1.0.0 的 JSON 报告 `schemaVersion` 仍为 `0.1`;工具版本与 schema 版本是两个独立兼容性维度。 +1.1.0 的 JSON 报告 `schemaVersion` 仍为 `0.1`;工具版本与 schema 版本是两个独立兼容性维度。 ```json { "schemaVersion": "0.1", - "tool": { "name": "skillbench", "version": "1.0.0" }, + "tool": { "name": "skillbench", "version": "1.1.0" }, "target": "/repo/SKILL.md", "score": { "overall": 91.4, "categories": {} }, "summary": { "info": 0, "warning": 2, "error": 0, "critical": 0 }, @@ -296,7 +296,7 @@ jobs: node-version: 20 package-manager-cache: false - name: Check agent instructions - run: npx --yes skillbench-ai@1.0.0 scan "$GITHUB_WORKSPACE" --ci --fail-on critical + run: npx --yes skillbench-ai@1.1.0 scan "$GITHUB_WORKSPACE" --ci --fail-on critical ``` 完整 SARIF / Code Scanning 示例见 [`examples/github-actions/skillbench-sarif.yml`](examples/github-actions/skillbench-sarif.yml)。 @@ -311,7 +311,7 @@ jobs: - **Performance Guard**:确定性 120 文件仓库规模测试; - **Publish preflight**:Node 24 + npm 11.18.0 再跑完整 `pnpm release:check`。 -发布流程先验证 npm Registry,再创建 `v1.0.0` 与 GitHub Release;不会用“先打 tag、后发现 npm 发布失败”的半发布状态冒充正式版本。完整标准见 [docs/1.0-RELEASE-CRITERIA.md](docs/1.0-RELEASE-CRITERIA.md),维护者流程见 [RELEASING.md](RELEASING.md)。 +发布流程先验证 npm Registry,再创建 `v1.1.0` 与 GitHub Release;不会用“先打 tag、后发现 npm 发布失败”的半发布状态冒充正式版本。完整标准见 [docs/1.0-RELEASE-CRITERIA.md](docs/1.0-RELEASE-CRITERIA.md),维护者流程见 [RELEASING.md](RELEASING.md)。 ## 开发 diff --git a/README_EN.md b/README_EN.md index 22e6ede..d34f194 100644 --- a/README_EN.md +++ b/README_EN.md @@ -14,14 +14,14 @@ CI MIT license Node 20 or newer - SkillBench version 1.0.0 + SkillBench version 1.1.0

![SkillBench terminal demo](docs/demo.svg) SkillBench 1.0 brings Agent Skills, `AGENTS.md`, `CLAUDE.md`, Cursor Rules, and similar instruction files into a repeatable engineering quality system: static checks, security rules, token efficiency, cross-agent compatibility, regression comparison, SARIF, CI gates, auditable safe fixes, and a human-labeled rule benchmark. -> **Release-integrity guarantee:** `v1.0.0` is created only after `skillbench-ai@1.0.0` has been successfully published and verified from the npm registry. If you are viewing an untagged release candidate, use the source workflow below. +> **Release-integrity guarantee:** `v1.1.0` is created only after `skillbench-ai@1.1.0` has been successfully published and verified from the npm registry. If you are viewing an untagged release candidate, use the source workflow below. ## Quick start @@ -30,13 +30,13 @@ SkillBench 1.0 brings Agent Skills, `AGENTS.md`, `CLAUDE.md`, Cursor Rules, and Run without a global install: ```bash -npx --yes skillbench-ai@1.0.0 scan SKILL.md +npx --yes skillbench-ai@1.1.0 scan SKILL.md ``` Or install the CLI globally: ```bash -npm install --global skillbench-ai@1.0.0 +npm install --global skillbench-ai@1.1.0 skillbench scan SKILL.md ``` @@ -251,12 +251,12 @@ Unknown keys, unknown rule IDs, and invalid severities fail explicitly rather th ## JSON / SARIF / API stability -SkillBench 1.0.0 keeps JSON report `schemaVersion` at `0.1`; tool version and report schema version are separate compatibility dimensions. +SkillBench 1.1.0 keeps JSON report `schemaVersion` at `0.1`; tool version and report schema version are separate compatibility dimensions. ```json { "schemaVersion": "0.1", - "tool": { "name": "skillbench", "version": "1.0.0" }, + "tool": { "name": "skillbench", "version": "1.1.0" }, "target": "/repo/SKILL.md", "score": { "overall": 91.4, "categories": {} }, "summary": { "info": 0, "warning": 2, "error": 0, "critical": 0 }, @@ -296,7 +296,7 @@ jobs: node-version: 20 package-manager-cache: false - name: Check agent instructions - run: npx --yes skillbench-ai@1.0.0 scan "$GITHUB_WORKSPACE" --ci --fail-on critical + run: npx --yes skillbench-ai@1.1.0 scan "$GITHUB_WORKSPACE" --ci --fail-on critical ``` A complete SARIF / Code Scanning example is available at [`examples/github-actions/skillbench-sarif.yml`](examples/github-actions/skillbench-sarif.yml). @@ -311,7 +311,7 @@ The exact 1.0 candidate must pass on the same commit: - **Performance Guard:** deterministic 120-file repository-scale workload; - **Publish preflight:** Node 24 + npm 11.18.0 running the complete `pnpm release:check` again. -The publish sequence verifies npm registry state before creating `v1.0.0` or the GitHub Release, preventing a failed npm publication from being presented as a completed release. See [docs/1.0-RELEASE-CRITERIA.md](docs/1.0-RELEASE-CRITERIA.md) and [RELEASING.md](RELEASING.md). +The publish sequence verifies npm registry state before creating `v1.1.0` or the GitHub Release, preventing a failed npm publication from being presented as a completed release. See [docs/1.0-RELEASE-CRITERIA.md](docs/1.0-RELEASE-CRITERIA.md) and [RELEASING.md](RELEASING.md). ## Development diff --git a/RELEASING.md b/RELEASING.md index 47c32d9..613e60e 100644 --- a/RELEASING.md +++ b/RELEASING.md @@ -61,7 +61,7 @@ The release gate fails on known high or critical advisories in production depend `.github/workflows/publish.yml` is the only repository workflow authorized to publish SkillBench. Pull requests that touch release metadata run a read-only Node.js 24 preflight with npm 11.18.0 and the complete `pnpm release:check`; they cannot publish packages or create releases. -Publishing jobs run only from the guarded 1.0 bootstrap on `main` or from version tags. They use a GitHub-hosted runner with `id-token: write`, npm provenance, the complete release gate, and a registry verification step before GitHub release creation. The workflow verifies that existing registry metadata points back to the canonical `EpochTX/skillbench` repository before treating an already-published version as an idempotent retry. +Publishing jobs run only from a validated version bump on `main` or from version tags. They use a GitHub-hosted runner with `id-token: write`, npm provenance, the complete release gate, and a registry verification step before GitHub release creation. The workflow verifies that existing registry metadata points back to the canonical `EpochTX/skillbench` repository before treating an already-published version as an idempotent retry. The publish workflow serializes main and tag runs so a newly created tag cannot race the main release. The first public `skillbench-ai` publication is a bootstrap case because npm requires the package to exist before a trusted publisher can be configured. The bootstrap publish therefore supports the repository `NPM_TOKEN` secret as a traditional-authentication fallback. npm's CLI prefers an available OIDC trusted-publisher identity before falling back to that token. After the first successful publication, configure npm trusted publishing for `EpochTX/skillbench` and `publish.yml`, verify the next OIDC publication, then revoke and remove the bootstrap write token. @@ -71,7 +71,7 @@ Do not commit npm tokens or long-lived publishing credentials to repository file Only release a commit that passed the full release gate and contains the exact released version metadata. -The 1.0 bootstrap publishes npm first and creates `v1.0.0` plus the GitHub release only after registry verification succeeds. Subsequent releases normally enter through a `vX.Y.Z` tag; the publish workflow verifies that the tag matches `package.json`, publishes or safely recognizes an idempotent matching registry version, verifies registry metadata, and creates the GitHub release if it does not already exist. +A version bump merged to `main` publishes npm first, verifies the registry and installed CLI, then creates `vX.Y.Z` and the GitHub release at that exact main commit. If that version and tag already exist, later main pushes skip publication. A manually created version tag can also invoke the idempotent tag workflow, which verifies that the tag matches `package.json` before publishing. Do not tag an `Unreleased` working state as an existing release version. diff --git a/docs/demo.svg b/docs/demo.svg index 4a2d4da..96e847d 100644 --- a/docs/demo.svg +++ b/docs/demo.svg @@ -10,7 +10,7 @@ $ npx skillbench-ai scan SKILL.md - SkillBench 1.0.0 + SkillBench 1.1.0 1 file scanned Overall Score 91.4 / 100 diff --git a/package.json b/package.json index 436bac1..271050e 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "skillbench-ai", - "version": "1.0.0", + "version": "1.1.0", "description": "Lint, score, secure and benchmark AI Agent Skills and instruction files across Codex, Claude Code, Cursor, Gemini CLI and GitHub Copilot.", "type": "module", "main": "./dist/index.js", diff --git a/src/core/token-analyzer.ts b/src/core/token-analyzer.ts index a0686a0..e414983 100644 --- a/src/core/token-analyzer.ts +++ b/src/core/token-analyzer.ts @@ -52,43 +52,45 @@ export function estimateTokens(value: string): number { } export function findDuplicateParagraphs(paragraphs: Paragraph[]): DuplicateMatch[] { - const candidates = paragraphs - .filter((paragraph) => normalizeText(paragraph.text).length >= 48) - .slice(0, 500) - .map((paragraph) => ({ - paragraph, - normalized: normalizeText(paragraph.text), - grams: ngrams(paragraph.text), - })); const matches: DuplicateMatch[] = []; - const alreadyMarked = new Set(); + const exactOriginals = new Map(); + const nearCandidates: { + paragraph: Paragraph; + normalized: string; + grams: Set; + }[] = []; - for (let leftIndex = 0; leftIndex < candidates.length; leftIndex += 1) { - const left = candidates[leftIndex]; - if (!left) continue; - for ( - let rightIndex = leftIndex + 1; - rightIndex < candidates.length; - rightIndex += 1 - ) { - const right = candidates[rightIndex]; - if (!right || alreadyMarked.has(right.paragraph.startLine)) continue; + for (const paragraph of paragraphs) { + const normalized = normalizeText(paragraph.text); + if (normalized.length < 48) continue; + + // Exact matches stay cheap and complete even in documents with >500 paragraphs. + const exactOriginal = exactOriginals.get(normalized); + if (exactOriginal) { + matches.push({ original: exactOriginal, duplicate: paragraph, similarity: 1 }); + continue; + } + exactOriginals.set(normalized, paragraph); + + // Fuzzy comparisons remain capped to avoid quadratic work on long inputs. + if (nearCandidates.length >= 500) continue; + const grams = ngrams(paragraph.text); + for (const left of nearCandidates) { const lengthRatio = - Math.min(left.normalized.length, right.normalized.length) / - Math.max(left.normalized.length, right.normalized.length); + Math.min(left.normalized.length, normalized.length) / + Math.max(left.normalized.length, normalized.length); if (lengthRatio < 0.72) continue; - - const similarity = - left.normalized === right.normalized ? 1 : jaccard(left.grams, right.grams); + const similarity = jaccard(left.grams, grams); if (similarity >= 0.86) { matches.push({ original: left.paragraph, - duplicate: right.paragraph, + duplicate: paragraph, similarity: round(similarity, 3), }); - alreadyMarked.add(right.paragraph.startLine); + break; } } + nearCandidates.push({ paragraph, normalized, grams }); } return matches; } diff --git a/src/version.ts b/src/version.ts index 11729f2..164b94e 100644 --- a/src/version.ts +++ b/src/version.ts @@ -1,3 +1,3 @@ -export const VERSION = '1.0.0'; +export const VERSION = '1.1.0'; export const SCHEMA_VERSION = '0.1' as const; export const PROJECT_URL = 'https://github.com/EpochTX/skillbench'; diff --git a/tests/unit/package-metadata.test.ts b/tests/unit/package-metadata.test.ts index 1890e70..29ab2ed 100644 --- a/tests/unit/package-metadata.test.ts +++ b/tests/unit/package-metadata.test.ts @@ -3,6 +3,8 @@ import path from 'node:path'; import { describe, expect, it } from 'vitest'; +import { VERSION } from '../../src/version.js'; + const root = path.resolve(import.meta.dirname, '../..'); const packageJson = JSON.parse( readFileSync(path.join(root, 'package.json'), 'utf8'), @@ -53,9 +55,9 @@ describe('package metadata', () => { expect(packageJson.scripts.prepublishOnly).toContain('pnpm verify'); }); - it('advertises the versioned npm command in the 1.0 terminal demo', () => { + it('advertises the current version in the terminal demo', () => { const demo = readFileSync(path.join(root, 'docs/demo.svg'), 'utf8'); expect(demo).toContain('npx skillbench-ai scan SKILL.md'); - expect(demo).toContain('SkillBench 1.0.0'); + expect(demo).toContain(`SkillBench ${VERSION}`); }); }); diff --git a/tests/unit/public-api.test.ts b/tests/unit/public-api.test.ts index 227d2dd..8032740 100644 --- a/tests/unit/public-api.test.ts +++ b/tests/unit/public-api.test.ts @@ -39,7 +39,7 @@ describe('public API', () => { it('keeps core registries and metadata canonical', () => { expect(api.builtInRules).toHaveLength(24); expect(api.builtInAdapters).toHaveLength(5); - expect(api.VERSION).toBe('1.0.0'); + expect(api.VERSION).toBe('1.1.0'); expect(api.SCHEMA_VERSION).toBe('0.1'); expect(api.PROJECT_URL).toBe('https://github.com/EpochTX/skillbench'); }); diff --git a/tests/unit/publish-workflow.test.ts b/tests/unit/publish-workflow.test.ts index 7e51498..9c73c67 100644 --- a/tests/unit/publish-workflow.test.ts +++ b/tests/unit/publish-workflow.test.ts @@ -18,10 +18,11 @@ describe('publish workflow', () => { expect(workflow).toContain('pnpm release:check'); }); - it('publishes only from the guarded main bootstrap or version tags', () => { + it('publishes from a validated main version bump or version tag', () => { expect(workflow).toContain("github.ref == 'refs/heads/main'"); expect(workflow).toContain("startsWith(github.ref, 'refs/tags/v')"); - expect(workflow).toContain("version\" != '1.0.0'"); + expect(workflow).toContain('release_sha="$GITHUB_SHA"'); + expect(workflow).toContain('echo \'eligible=false\' >> "$GITHUB_OUTPUT"'); expect(workflow).toContain('npm publish --access public --provenance'); expect(workflow).toContain('NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }}'); }); diff --git a/tests/unit/token-analyzer.test.ts b/tests/unit/token-analyzer.test.ts index f726581..23c925b 100644 --- a/tests/unit/token-analyzer.test.ts +++ b/tests/unit/token-analyzer.test.ts @@ -3,6 +3,7 @@ import { describe, expect, it } from 'vitest'; import { TokenEfficiencyAnalyzer, estimateTokens, + findDuplicateParagraphs, } from '../../src/core/token-analyzer.js'; import { parseDocument } from '../../src/parser/parser.js'; @@ -24,4 +25,19 @@ describe('TokenEfficiencyAnalyzer', () => { expect(metrics.duplicateTokens).toBeGreaterThan(0); expect(metrics.redundancyRatio).toBeGreaterThan(0.3); }); + + it('finds exact repetition beyond the fuzzy-comparison limit', () => { + const paragraphs = Array.from({ length: 550 }, (_, index) => ({ + text: `Inspect the independent instruction number ${index} and record its evidence before moving to the next one.`, + startLine: index * 2 + 1, + endLine: index * 2 + 1, + })); + paragraphs.push({ ...paragraphs[525]!, startLine: 1101, endLine: 1101 }); + const matches = findDuplicateParagraphs(paragraphs); + expect(matches).toContainEqual({ + original: paragraphs[525], + duplicate: paragraphs[550], + similarity: 1, + }); + }); });