diff --git a/.github/workflows/autoloop.lock.yml b/.github/workflows/autoloop.lock.yml index 0683ab676..08efb1490 100644 --- a/.github/workflows/autoloop.lock.yml +++ b/.github/workflows/autoloop.lock.yml @@ -1,4 +1,4 @@ -# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"cabe05311ee3662402279684a00f97aa7f21d8392fa46b6032c3472d54afbdf5","body_hash":"73645386e4db5f30cf7d7de628df076045e04a2d38b5c415fd4fd22ba4e78ad0","compiler_version":"v0.87.10","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.80"}} +# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"14909602d87a29c752068070b44c3f6801e214257e11466cbf4de9e3ab9687c0","body_hash":"e27f2f23448b83d1d291e0d02daef00c18be34073bcfcd8c5dd2f1e58ee6a334","compiler_version":"v0.87.10","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.80"}} # gh-aw-manifest: {"version":1,"secrets":["COPILOT_GITHUB_TOKEN","GH_AW_CI_TRIGGER_TOKEN","GH_AW_DEFAULT_OTLP_HEADERS","GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GITHUB_TOKEN"],"actions":[{"repo":"actions/cache/restore","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/save","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/checkout","sha":"3d3c42e5aac5ba805825da76410c181273ba90b1","version":"v7.0.1"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/setup-python","sha":"5fda3b95a4ea91299a34e894583c3862153e4b97","version":"v7.0.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"},{"repo":"github/gh-aw-actions/setup","sha":"bc8c008a419c5b7a29df6f5641edd35fd1c6ea85","version":"v0.87.10"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.28.10","digest":"sha256:c01e6d16d11ea4f2a46cc023a9f402224a3b3861b026818eec0dc586d7e6918e","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.28.10@sha256:c01e6d16d11ea4f2a46cc023a9f402224a3b3861b026818eec0dc586d7e6918e"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.10","digest":"sha256:c3a18aebb8251339117ea998296315de17bada366f8d03919b3348ea71112e64","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.10@sha256:c3a18aebb8251339117ea998296315de17bada366f8d03919b3348ea71112e64"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.28.10","digest":"sha256:c06076f7aca95df713e0748c44d80c0a3c2538fad67bfdd04296d45158e083e6","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.28.10@sha256:c06076f7aca95df713e0748c44d80c0a3c2538fad67bfdd04296d45158e083e6"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.14","digest":"sha256:b2f0c2b2f17b5fbe809e5bb99dc185b6ddd70df25295dc63a6d526350334eff5","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.14@sha256:b2f0c2b2f17b5fbe809e5bb99dc185b6ddd70df25295dc63a6d526350334eff5"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:bac2192f6374d6262116399b34fc5e143d576f82719e90a18261cae7480f4d4e","pinned_image":"ghcr.io/github/gh-aw-node@sha256:bac2192f6374d6262116399b34fc5e143d576f82719e90a18261cae7480f4d4e"},{"image":"ghcr.io/github/github-mcp-server:v1.11.0","digest":"sha256:fbec75de11c255213fa08d80fb166abe73d851fff631c51c0079872967720699","pinned_image":"ghcr.io/github/github-mcp-server:v1.11.0@sha256:fbec75de11c255213fa08d80fb166abe73d851fff631c51c0079872967720699"}],"mcp_servers":[{"name":"github","tools":["actions_get","actions_list","get_code_scanning_alert","get_commit","get_discussion","get_discussion_comments","get_file_contents","get_job_logs","get_label","get_latest_release","get_me","get_notification_details","get_pull_request","get_pull_request_comments","get_pull_request_diff","get_pull_request_files","get_pull_request_review_comments","get_pull_request_reviews","get_pull_request_status","get_release_by_tag","get_secret_scanning_alert","get_tag","issue_read","list_branches","list_code_scanning_alerts","list_commits","list_discussion_categories","list_discussions","list_issue_types","list_issues","list_label","list_notifications","list_pull_requests","list_releases","list_secret_scanning_alerts","list_starred_repositories","list_tags","pull_request_read","search_code","search_issues","search_orgs","search_pull_requests","search_repositories","search_users"]},{"name":"safeoutputs","tools":["add_comment","add_labels","create_issue","create_pull_request","missing_data","missing_tool","noop","push_to_pull_request_branch","remove_labels","update_issue"]}]} # This file was automatically generated by gh-aw (v0.87.10). DO NOT EDIT. To debug this workflow, load the skill at https://github.com/github/gh-aw/blob/main/debug.md # @@ -425,7 +425,7 @@ jobs: GH_AW_INCLUDE_PR_CONTEXT: ${{ (github.event_name == 'issue_comment' && github.event.issue.pull_request != null) || github.event_name == 'pull_request_review_comment' || github.event_name == 'pull_request_review' }} GH_AW_MCP_CLI_SERVERS_LIST: "- `github` — run `github --help` to see available tools\n- `safeoutputs` — run `safeoutputs --help` to see available tools" GH_AW_MEMORY_BRANCH_NAME: 'memory/autoloop' - GH_AW_MEMORY_CONSTRAINTS: "\n\n**Constraints:**\n- **Allowed Files**: Only files matching patterns: *.md\n- **Max File Size**: 30720 bytes (0.03 MB) per file\n- **Max File Count**: 100 files per commit\n- **Max Patch Size**: 10240 bytes (10 KB) total per push (max: 1024 KB)\n" + GH_AW_MEMORY_CONSTRAINTS: "\n\n**Constraints:**\n- **Max File Size**: 30720 bytes (0.03 MB) per file\n- **Max File Count**: 100 files per commit\n- **Max Patch Size**: 10240 bytes (10 KB) total per push (max: 1024 KB)\n" GH_AW_MEMORY_DESCRIPTION: '' GH_AW_MEMORY_DIR: '/tmp/gh-aw/repo-memory/default/' GH_AW_MEMORY_TARGET_REPO: ' of the current repository' @@ -687,7 +687,7 @@ jobs: env: GH_AW_FILE_ROOT: "${{ runner.temp }}/gh-aw" GH_AW_FILE_CONFIG: "{\"files\":[{\"path\":\"safeoutputs/config.json\",\"content_env\":\"GH_AW_SAFE_OUTPUTS_CONFIG\"}]}" - GH_AW_SAFE_OUTPUTS_CONFIG: "{\"add_comment\":{\"hide_older_comments\":false,\"max\":7,\"target\":\"*\"},\"add_labels\":{\"max\":2,\"target\":\"*\"},\"create_issue\":{\"labels\":[\"automation\",\"autoloop\"],\"max\":1},\"create_pull_request\":{\"draft\":true,\"labels\":[\"automation\",\"autoloop\"],\"max\":1,\"max_patch_files\":500,\"max_patch_size\":10240,\"preserve_branch_name\":true,\"protect_top_level_dot_folders\":true,\"protected_files\":[\"deno.json\",\"deno.jsonc\",\"deno.lock\",\"global.json\",\"NuGet.Config\",\"Directory.Packages.props\",\"mix.exs\",\"mix.lock\",\"go.mod\",\"go.sum\",\"stack.yaml\",\"stack.yaml.lock\",\"pom.xml\",\"build.gradle\",\"build.gradle.kts\",\"settings.gradle\",\"settings.gradle.kts\",\"gradle.properties\",\"npm-shrinkwrap.json\",\"Pipfile\",\"Pipfile.lock\",\"Gemfile\",\"Gemfile.lock\",\"uv.lock\",\"CODEOWNERS\",\"DESIGN.md\",\"README.md\",\"CONTRIBUTING.md\",\"CHANGELOG.md\",\"SECURITY.md\",\"CODE_OF_CONDUCT.md\",\"AGENTS.md\"],\"protected_files_policy\":\"fallback-to-issue\",\"recreate_ref\":true},\"create_report_incomplete_issue\":{},\"missing_data\":{},\"missing_tool\":{},\"noop\":{\"max\":1,\"report-as-issue\":\"false\"},\"push_repo_memory\":{\"memories\":[{\"dir\":\"/tmp/gh-aw/repo-memory/default\",\"id\":\"default\",\"max_file_count\":100,\"max_file_size\":30720,\"max_patch_size\":10240}]},\"push_to_pull_request_branch\":{\"if_no_changes\":\"warn\",\"max\":1,\"max_patch_size\":10240,\"protect_top_level_dot_folders\":true,\"protected_files\":[\"deno.json\",\"deno.jsonc\",\"deno.lock\",\"global.json\",\"NuGet.Config\",\"Directory.Packages.props\",\"mix.exs\",\"mix.lock\",\"go.mod\",\"go.sum\",\"stack.yaml\",\"stack.yaml.lock\",\"pom.xml\",\"build.gradle\",\"build.gradle.kts\",\"settings.gradle\",\"settings.gradle.kts\",\"gradle.properties\",\"npm-shrinkwrap.json\",\"Pipfile\",\"Pipfile.lock\",\"Gemfile\",\"Gemfile.lock\",\"uv.lock\",\"CODEOWNERS\",\"DESIGN.md\",\"README.md\",\"CONTRIBUTING.md\",\"CHANGELOG.md\",\"SECURITY.md\",\"CODE_OF_CONDUCT.md\",\"AGENTS.md\"],\"protected_files_policy\":\"allowed\",\"signed_commits\":false,\"target\":\"*\",\"title_prefix\":\"[Autoloop\"},\"remove_labels\":{\"max\":2,\"target\":\"*\"},\"report_incomplete\":{},\"update_issue\":{\"allow_body\":true,\"max\":3,\"target\":\"*\",\"title_prefix\":\"[Autoloop\"}}" + GH_AW_SAFE_OUTPUTS_CONFIG: "{\"add_comment\":{\"hide_older_comments\":false,\"max\":7,\"target\":\"*\"},\"add_labels\":{\"max\":2,\"target\":\"*\"},\"create_issue\":{\"labels\":[\"automation\",\"autoloop\"],\"max\":1},\"create_pull_request\":{\"draft\":true,\"labels\":[\"automation\",\"autoloop\"],\"max\":1,\"max_patch_files\":64,\"max_patch_size\":10240,\"preserve_branch_name\":true,\"protect_top_level_dot_folders\":true,\"protected_files\":[\"deno.json\",\"deno.jsonc\",\"deno.lock\",\"global.json\",\"NuGet.Config\",\"Directory.Packages.props\",\"mix.exs\",\"mix.lock\",\"go.mod\",\"go.sum\",\"stack.yaml\",\"stack.yaml.lock\",\"pom.xml\",\"build.gradle\",\"build.gradle.kts\",\"settings.gradle\",\"settings.gradle.kts\",\"gradle.properties\",\"npm-shrinkwrap.json\",\"Pipfile\",\"Pipfile.lock\",\"Gemfile\",\"Gemfile.lock\",\"uv.lock\",\"CODEOWNERS\",\"DESIGN.md\",\"README.md\",\"CONTRIBUTING.md\",\"CHANGELOG.md\",\"SECURITY.md\",\"CODE_OF_CONDUCT.md\",\"AGENTS.md\"],\"protected_files_policy\":\"fallback-to-issue\",\"recreate_ref\":true},\"create_report_incomplete_issue\":{},\"missing_data\":{},\"missing_tool\":{},\"noop\":{\"max\":1,\"report-as-issue\":\"false\"},\"push_repo_memory\":{\"memories\":[{\"dir\":\"/tmp/gh-aw/repo-memory/default\",\"id\":\"default\",\"max_file_count\":100,\"max_file_size\":30720,\"max_patch_size\":10240}]},\"push_to_pull_request_branch\":{\"if_no_changes\":\"warn\",\"max\":1,\"max_patch_size\":10240,\"protect_top_level_dot_folders\":true,\"protected_files\":[\"deno.json\",\"deno.jsonc\",\"deno.lock\",\"global.json\",\"NuGet.Config\",\"Directory.Packages.props\",\"mix.exs\",\"mix.lock\",\"go.mod\",\"go.sum\",\"stack.yaml\",\"stack.yaml.lock\",\"pom.xml\",\"build.gradle\",\"build.gradle.kts\",\"settings.gradle\",\"settings.gradle.kts\",\"gradle.properties\",\"npm-shrinkwrap.json\",\"Pipfile\",\"Pipfile.lock\",\"Gemfile\",\"Gemfile.lock\",\"uv.lock\",\"CODEOWNERS\",\"DESIGN.md\",\"README.md\",\"CONTRIBUTING.md\",\"CHANGELOG.md\",\"SECURITY.md\",\"CODE_OF_CONDUCT.md\",\"AGENTS.md\"],\"protected_files_policy\":\"allowed\",\"signed_commits\":false,\"target\":\"*\",\"title_prefix\":\"[Autoloop\"},\"remove_labels\":{\"max\":2,\"target\":\"*\"},\"report_incomplete\":{},\"update_issue\":{\"allow_body\":true,\"max\":3,\"target\":\"*\",\"title_prefix\":\"[Autoloop\"}}" with: script: | const path = require('path'); @@ -2191,8 +2191,7 @@ jobs: MAX_FILE_SIZE: 30720 MAX_FILE_COUNT: 100 MAX_PATCH_SIZE: 10240 - ALLOWED_EXTENSIONS: '[]' - FILE_GLOB_FILTER: "*.md" + ALLOWED_EXTENSIONS: '[".md"]' with: script: | const path = require('path'); @@ -2333,7 +2332,7 @@ jobs: GH_AW_ALLOWED_DOMAINS: "*.gradle-enterprise.cloud,*.pythonhosted.org,*.vsblob.vsassets.io,adoptium.net,anaconda.org,api.adoptium.net,api.foojay.io,api.npms.io,api.nuget.org,api.snapcraft.io,archive.apache.org,archive.ubuntu.com,azure.archive.ubuntu.com,azuresearch-usnc.nuget.org,azuresearch-ussc.nuget.org,binstar.org,bootstrap.pypa.io,builds.dotnet.microsoft.com,bun.sh,cdn.azul.com,cdn.jsdelivr.net,central.sonatype.com,ci.dot.net,conda.anaconda.org,conda.binstar.org,crates.io,crl.geotrust.com,crl.globalsign.com,crl.identrust.com,crl.sectigo.com,crl.thawte.com,crl.usertrust.com,crl.verisign.com,crl3.digicert.com,crl4.digicert.com,crls.ssl.com,dc.services.visualstudio.com,deb.nodesource.com,deno.land,develocity.apache.org,dist.nuget.org,dl.google.com,dlcdn.apache.org,dot.net,dotnet.microsoft.com,dotnetcli.blob.core.windows.net,download.eclipse.org,download.java.net,download.oracle.com,downloads.gradle-dn.com,esm.sh,files.pythonhosted.org,ge.spockframework.org,get.pnpm.io,googleapis.deno.dev,googlechromelabs.github.io,gradle.org,index.crates.io,jcenter.bintray.com,jdk.java.net,json-schema.org,json.schemastore.org,jsr.io,keyserver.ubuntu.com,maven-central.storage-download.googleapis.com,maven.apache.org,maven.google.com,maven.oracle.com,maven.pkg.github.com,nodejs.org,npm.pkg.github.com,npmjs.com,npmjs.org,nuget.org,nuget.pkg.github.com,nugetregistryv2prod.blob.core.windows.net,ocsp.digicert.com,ocsp.geotrust.com,ocsp.globalsign.com,ocsp.identrust.com,ocsp.sectigo.com,ocsp.ssl.com,ocsp.thawte.com,ocsp.usertrust.com,ocsp.verisign.com,oneocsp.microsoft.com,packagecloud.io,packages.cloud.google.com,packages.microsoft.com,pip.pypa.io,pkgs.dev.azure.com,plugins-artifacts.gradle.org,plugins.gradle.org,ppa.launchpad.net,pypi.org,pypi.python.org,registry.bower.io,registry.npmjs.com,registry.npmjs.org,registry.yarnpkg.com,repo.anaconda.com,repo.continuum.io,repo.gradle.org,repo.grails.org,repo.maven.apache.org,repo.spring.io,repo.yarnpkg.com,repo1.maven.org,repository.apache.org,s.symcb.com,s.symcd.com,scans-in.gradle.com,security.ubuntu.com,services.gradle.org,sh.rustup.rs,skimdb.npmjs.com,static.crates.io,static.rust-lang.org,storage.googleapis.com,telemetry.vercel.com,ts-crl.ws.symantec.com,ts-ocsp.ws.symantec.com,www.googleapis.com,www.java.com,www.microsoft.com,www.npmjs.com,www.npmjs.org,yarnpkg.com" GITHUB_SERVER_URL: ${{ github.server_url }} GITHUB_API_URL: ${{ github.api_url }} - GH_AW_SAFE_OUTPUTS_HANDLER_CONFIG: "{\"add_comment\":{\"hide_older_comments\":false,\"max\":7,\"target\":\"*\"},\"add_labels\":{\"max\":2,\"target\":\"*\"},\"create_issue\":{\"labels\":[\"automation\",\"autoloop\"],\"max\":1},\"create_pull_request\":{\"draft\":true,\"labels\":[\"automation\",\"autoloop\"],\"max\":1,\"max_patch_files\":500,\"max_patch_size\":10240,\"preserve_branch_name\":true,\"protect_top_level_dot_folders\":true,\"protected_files\":[\"deno.json\",\"deno.jsonc\",\"deno.lock\",\"global.json\",\"NuGet.Config\",\"Directory.Packages.props\",\"mix.exs\",\"mix.lock\",\"go.mod\",\"go.sum\",\"stack.yaml\",\"stack.yaml.lock\",\"pom.xml\",\"build.gradle\",\"build.gradle.kts\",\"settings.gradle\",\"settings.gradle.kts\",\"gradle.properties\",\"npm-shrinkwrap.json\",\"Pipfile\",\"Pipfile.lock\",\"Gemfile\",\"Gemfile.lock\",\"uv.lock\",\"CODEOWNERS\",\"DESIGN.md\",\"README.md\",\"CONTRIBUTING.md\",\"CHANGELOG.md\",\"SECURITY.md\",\"CODE_OF_CONDUCT.md\",\"AGENTS.md\"],\"protected_files_policy\":\"fallback-to-issue\",\"recreate_ref\":true},\"create_report_incomplete_issue\":{},\"missing_data\":{},\"missing_tool\":{},\"noop\":{\"max\":1,\"report-as-issue\":\"false\"},\"push_to_pull_request_branch\":{\"if_no_changes\":\"warn\",\"max\":1,\"max_patch_size\":10240,\"protect_top_level_dot_folders\":true,\"protected_files\":[\"deno.json\",\"deno.jsonc\",\"deno.lock\",\"global.json\",\"NuGet.Config\",\"Directory.Packages.props\",\"mix.exs\",\"mix.lock\",\"go.mod\",\"go.sum\",\"stack.yaml\",\"stack.yaml.lock\",\"pom.xml\",\"build.gradle\",\"build.gradle.kts\",\"settings.gradle\",\"settings.gradle.kts\",\"gradle.properties\",\"npm-shrinkwrap.json\",\"Pipfile\",\"Pipfile.lock\",\"Gemfile\",\"Gemfile.lock\",\"uv.lock\",\"CODEOWNERS\",\"DESIGN.md\",\"README.md\",\"CONTRIBUTING.md\",\"CHANGELOG.md\",\"SECURITY.md\",\"CODE_OF_CONDUCT.md\",\"AGENTS.md\"],\"protected_files_policy\":\"allowed\",\"signed_commits\":false,\"target\":\"*\",\"title_prefix\":\"[Autoloop\"},\"remove_labels\":{\"max\":2,\"target\":\"*\"},\"report_incomplete\":{},\"update_issue\":{\"allow_body\":true,\"max\":3,\"target\":\"*\",\"title_prefix\":\"[Autoloop\"}}" + GH_AW_SAFE_OUTPUTS_HANDLER_CONFIG: "{\"add_comment\":{\"hide_older_comments\":false,\"max\":7,\"target\":\"*\"},\"add_labels\":{\"max\":2,\"target\":\"*\"},\"create_issue\":{\"labels\":[\"automation\",\"autoloop\"],\"max\":1},\"create_pull_request\":{\"draft\":true,\"labels\":[\"automation\",\"autoloop\"],\"max\":1,\"max_patch_files\":64,\"max_patch_size\":10240,\"preserve_branch_name\":true,\"protect_top_level_dot_folders\":true,\"protected_files\":[\"deno.json\",\"deno.jsonc\",\"deno.lock\",\"global.json\",\"NuGet.Config\",\"Directory.Packages.props\",\"mix.exs\",\"mix.lock\",\"go.mod\",\"go.sum\",\"stack.yaml\",\"stack.yaml.lock\",\"pom.xml\",\"build.gradle\",\"build.gradle.kts\",\"settings.gradle\",\"settings.gradle.kts\",\"gradle.properties\",\"npm-shrinkwrap.json\",\"Pipfile\",\"Pipfile.lock\",\"Gemfile\",\"Gemfile.lock\",\"uv.lock\",\"CODEOWNERS\",\"DESIGN.md\",\"README.md\",\"CONTRIBUTING.md\",\"CHANGELOG.md\",\"SECURITY.md\",\"CODE_OF_CONDUCT.md\",\"AGENTS.md\"],\"protected_files_policy\":\"fallback-to-issue\",\"recreate_ref\":true},\"create_report_incomplete_issue\":{},\"missing_data\":{},\"missing_tool\":{},\"noop\":{\"max\":1,\"report-as-issue\":\"false\"},\"push_to_pull_request_branch\":{\"if_no_changes\":\"warn\",\"max\":1,\"max_patch_size\":10240,\"protect_top_level_dot_folders\":true,\"protected_files\":[\"deno.json\",\"deno.jsonc\",\"deno.lock\",\"global.json\",\"NuGet.Config\",\"Directory.Packages.props\",\"mix.exs\",\"mix.lock\",\"go.mod\",\"go.sum\",\"stack.yaml\",\"stack.yaml.lock\",\"pom.xml\",\"build.gradle\",\"build.gradle.kts\",\"settings.gradle\",\"settings.gradle.kts\",\"gradle.properties\",\"npm-shrinkwrap.json\",\"Pipfile\",\"Pipfile.lock\",\"Gemfile\",\"Gemfile.lock\",\"uv.lock\",\"CODEOWNERS\",\"DESIGN.md\",\"README.md\",\"CONTRIBUTING.md\",\"CHANGELOG.md\",\"SECURITY.md\",\"CODE_OF_CONDUCT.md\",\"AGENTS.md\"],\"protected_files_policy\":\"allowed\",\"signed_commits\":false,\"target\":\"*\",\"title_prefix\":\"[Autoloop\"},\"remove_labels\":{\"max\":2,\"target\":\"*\"},\"report_incomplete\":{},\"update_issue\":{\"allow_body\":true,\"max\":3,\"target\":\"*\",\"title_prefix\":\"[Autoloop\"}}" GH_AW_CI_TRIGGER_TOKEN: ${{ secrets.GH_AW_CI_TRIGGER_TOKEN }} with: github-token: ${{ secrets.GH_AW_GITHUB_TOKEN || secrets.GITHUB_TOKEN }} diff --git a/.github/workflows/autoloop.md b/.github/workflows/autoloop.md index f595fe03c..d472f55ee 100644 --- a/.github/workflows/autoloop.md +++ b/.github/workflows/autoloop.md @@ -36,7 +36,7 @@ network: safe-outputs: max-patch-size: 10240 - max-patch-files: 500 + max-patch-files: 64 add-comment: max: 7 target: "*" @@ -107,7 +107,8 @@ tools: bash: true repo-memory: branch-name: memory/autoloop - file-glob: ["*.md"] + # Slashless globs in gh-aw v0.87.10 exclude files at the memory root. + allowed-extensions: [".md"] # 30 KB per state file -- enough for the structured sections plus ~10 most-recent # iteration entries plus ~5 compressed-range summaries. The rolling-compaction # rule in "Update Rules" below keeps files under this budget. Tune up for @@ -137,6 +138,43 @@ features: An iterative optimization agent that proposes changes, evaluates them against a metric, and keeps only improvements — running autonomously on a schedule. +## Objective And Evidence Guard + +Apply this guard to every mode and program-specific strategy. The current +program definition and `AGENTS.md` define the objective and allowed scope; +repo-memory is working history, not permission to expand them. Recheck inherited +claims against the current remote branch. Retire stale priorities that reward +bulk domain generation, unrelated scientific modules, duplicate files, or stubs. +Choose one small, useful checkpoint toward the actual program goal. + +- **Pandas parity:** name the pandas API and supported behavior being added or + repaired. Tests must execute the tsb operation on independently specified + inputs and compare with independently generated pandas results, including + relevant values, labels, dtypes, missing values, and errors. Reconstructing a + pandas expected snapshot as a tsb container and comparing it back is not + differential evidence. A real fix in an existing file can matter more than a + new file; never manufacture files to make it score. +- **Performance:** execute each new or changed benchmark on both sides and + verify equivalent outputs before trusting timings. Match operation, dataset, + dtype, and warm-up/measurement policy; record versions, measured SHA, repeated + timings, and variability. Separate cold work from repeated-input cache hits; + do not claim sorting parity by comparing cached tsb returns with fresh pandas + sorts. Do not add import-time JIT primers solely to improve the benchmark. + Script counts or successful parsing do not prove execution or speedup. +- **Progress:** retain the program's declared metric, but report what it + actually measures. Do not promote file counts or benchmark-pair counts into + feature parity or performance completion. A completion percentage needs an + explicit in-scope denominator and a verified numerator, with unsupported and + unverified cases separate. If these are absent, report completeness as + unknown. Cite current-head verification, not stale memory totals. + +If the evaluator conflicts with the objective, report `metric-contract-mismatch` +with concrete evidence and the smallest proposed correction. Do not redefine +the metric, edit protected program definitions or issue #1, raise `best_metric`, +or claim completion. Preserve useful findings in memory without publishing +unrelated work. A prior inflated or unverified best is not a target to beat by +padding; ask for its evidence-based reset instead. + ## Command Mode Take heed of **instructions**: "${{ steps.sanitized.outputs.text }}" @@ -206,7 +244,9 @@ The pre-step has already determined which program to run. Read `/tmp/gh-aw/autol - **`selected_file`**: The full path to the program's markdown file (either `.autoloop/programs//program.md`, `.autoloop/programs/.md`, or `/tmp/gh-aw/issue-programs/.md` for issue-based programs). - **`selected_issue`**: The GitHub issue number if the selected program came from an issue, or `null` if it came from a file. - **`selected_target_metric`**: The `target-metric` value from the program's frontmatter (a number), or `null` if the program is open-ended. Used to check the [halting condition](#halting-condition) after each accepted iteration. -- **`selected_metric_direction`**: One of `"higher"` (default) or `"lower"`, parsed from the program's `metric_direction` frontmatter field. Determines whether **larger** or **smaller** metric values count as improvement. Used by the metric-improved check in [Step 5](#step-5-accept-or-reject), the iteration-history delta sign, and the [halting condition](#halting-condition). +- **`selected_metric_direction`**: Optional `"higher"` or `"lower"`. Validate against the program's contract using [Metric Direction](#metric-direction), including its missing-field fallback, before comparing or accepting results. +- **`selected_metric_direction_error`**: Contract ambiguity or conflict to report without accepting a metric. +- **`selected_reconciliation`**: `tree` for current pending publication, `legacy` for an unresolved older candidate, or `null`. Reconcile evidence before proposing work; this is never acceptance evidence itself. - **`state_file_size_bytes`**: Current size of the selected program's state file in bytes (0 if it does not exist yet). Use this together with `state_file_max_bytes` to decide whether to compact aggressively this iteration (see [Update Rules](#update-rules) — when size exceeds 80% of the max, collapse older iteration entries). - **`state_file_max_bytes`**: The configured `max-file-size` for repo-memory state files (default `30720`, i.e. 30 KB). Files larger than this are rejected by repo-memory, breaking scheduling. - **`issue_programs`**: A mapping of program name → issue number for all discovered issue-based programs. @@ -286,17 +326,18 @@ target-metric: 0.95 ### Metric Direction -By default Autoloop assumes **higher is better** — `best_metric` is ratcheted up each accepted iteration, and a `target-metric` is met when `best_metric >= target-metric`. Programs whose natural fitness is *lower is better* (error, latency, cost, ratio, fitness score) can opt into reversed semantics with the optional `metric_direction` field: +Resolve metric direction from the program's evaluation contract before comparing +results. The optional `metric_direction` frontmatter field makes it explicit: ```markdown --- schedule: every 6h -metric_direction: lower # defaults to "higher" if omitted +metric_direction: lower # smaller ratios are better target-metric: 0.9 # interpreted as "program is complete when best_metric ≤ 0.9" --- ``` -Allowed values are `higher` (default) and `lower`. Any other value is rejected at frontmatter-parse time, the scheduler logs a warning, and the program falls back to `higher`. +Allowed values are `higher` and `lower`. When `metric_direction: lower` is set: @@ -304,7 +345,12 @@ When `metric_direction: lower` is set: - Iteration History entries show a `-` (negative delta = improvement) instead of `+`. - The halting condition fires when `best_metric <= target-metric` (instead of `>=`). -The agent reads `selected_metric_direction` from `/tmp/gh-aw/autoloop.json` to determine which direction applies to the current iteration. Programs that omit the field are treated as `higher` — no behaviour change for existing programs. +Read `selected_metric_direction` from `/tmp/gh-aw/autoloop.json` when present and +check it against the program. If the scheduler field is missing, resolve the +explicit direction from the program frontmatter or Evaluation prose; for +example, `tsb-perf-evolve` minimizes its ratio. If direction is ambiguous or +conflicting, stop with `metric-contract-mismatch`. Never silently assume +`higher`, rely on stale memory, or change the protected definition to proceed. ## Program Definition @@ -394,6 +440,16 @@ durable state only in this directory; a separate clone is not persisted. If the machine state has a `Pending Tree`, reconcile it using Step 5b and end the run before starting Step 2. +When `selected_reconciliation` is `legacy`, reconcile the newest unresolved +population/history entry before proposing anything. Locate its recorded commit +and run; verify publication, current branch identity, metric, and CI rather than +trusting `pending-ci` text. Restore current pending fields only when that exact +candidate and evidence can be established. If it was already merged, superseded, +or cannot be verified, retire the stale pending marker with the reason; do not +accept its claimed metric. Update the authoritative memory and end this run. +An explicit pause or completion remains a stop; pending work only bypasses an +automatic rejection-plateau skip. + ### Step 2: Analyze and Propose 1. Read the target files and understand the current state. @@ -550,7 +606,7 @@ when `Pending Tree` is present. - Update **📚 Lessons Learned** if this iteration revealed something new about the problem or what works. - Update **🔭 Future Directions** if this iteration opened new promising paths. 6. **Update the program issue**: edit the status comment and post a per-iteration comment on the program issue (see [Program Issue](#program-issue)). Note the fix-attempt count in the per-iteration comment if `> 0`. -7. **Check halting condition** (see [Halting Condition](#halting-condition)): If the program has a `target-metric` in its frontmatter, compare the new `best_metric` against it using the program's metric direction (read `selected_metric_direction` from `/tmp/gh-aw/autoloop.json`): +7. **Check halting condition** (see [Halting Condition](#halting-condition)): If the program has a `target-metric` in its frontmatter, compare the new `best_metric` against it using the validated [Metric Direction](#metric-direction): - `higher`: completed when `best_metric >= target-metric`. - `lower`: completed when `best_metric <= target-metric`. @@ -665,8 +721,8 @@ Programs can be **open-ended** (run indefinitely until manually stopped) or **go 1. Parse the `target-metric` value from the program's YAML frontmatter (if present). 2. After each **accepted** iteration, compare the new `best_metric` against the `target-metric`. -3. Determine whether the target is met based on the program's `metric_direction` (read from `selected_metric_direction` in `/tmp/gh-aw/autoloop.json`; defaults to `higher` when unset): - - `higher` (default): the target is met when `best_metric >= target-metric`. +3. Determine whether the target is met using the validated [Metric Direction](#metric-direction), not an assumed scheduler default: + - `higher`: the target is met when `best_metric >= target-metric`. - `lower`: the target is met when `best_metric <= target-metric`. 4. When the target is met, **complete** the program: - Set `Completed` to `true` in the state file's **⚙️ Machine State** table. @@ -835,7 +891,7 @@ All iterations in reverse chronological order (newest first). | Iteration Count | integer | Total iterations completed | | Best Metric | number | Best metric value achieved so far | | Target Metric | number or `—` | Target metric from program frontmatter (halting condition). `—` if open-ended | -| Metric Direction | `higher` or `lower` | Whether larger or smaller metric values count as improvement. Defaults to `higher` if absent (back-compat). Set from the program's `metric_direction` frontmatter field. | +| Metric Direction | `higher` or `lower` | Validated against the current program's frontmatter or Evaluation contract; missing or conflicting evidence must not silently default. | | Branch | branch name | Long-running branch: `autoloop/{program-name}` | | PR | `#number` or `—` | Draft PR number for this program | | Issue | `#number` or `—` | The single program issue (`[Autoloop: {program-name}]`) for this program. Hosts the status comment, per-iteration comments, and human steering comments. | diff --git a/.github/workflows/goal.lock.yml b/.github/workflows/goal.lock.yml index ac51daff8..30c4bbbb2 100644 --- a/.github/workflows/goal.lock.yml +++ b/.github/workflows/goal.lock.yml @@ -1,5 +1,5 @@ -# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"4de5fc613741801bdd7226e7e5bf3c369016ca6f90b1e91639ce4d8361a03821","body_hash":"3ea5f188158645fe2d65d6f12b1615734798a6422a5f37d6b090f11ed4d233d9","compiler_version":"v0.87.10","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.80"}} -# gh-aw-manifest: {"version":1,"secrets":["COPILOT_GITHUB_TOKEN","GH_AW_CI_TRIGGER_TOKEN","GH_AW_DEFAULT_OTLP_HEADERS","GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GITHUB_TOKEN"],"actions":[{"repo":"actions/cache/restore","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/save","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/checkout","sha":"3d3c42e5aac5ba805825da76410c181273ba90b1","version":"v7.0.1"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/setup-python","sha":"5fda3b95a4ea91299a34e894583c3862153e4b97","version":"v7.0.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"},{"repo":"github/gh-aw-actions/setup","sha":"bc8c008a419c5b7a29df6f5641edd35fd1c6ea85","version":"v0.87.10"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.28.10","digest":"sha256:c01e6d16d11ea4f2a46cc023a9f402224a3b3861b026818eec0dc586d7e6918e","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.28.10@sha256:c01e6d16d11ea4f2a46cc023a9f402224a3b3861b026818eec0dc586d7e6918e"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.10","digest":"sha256:c3a18aebb8251339117ea998296315de17bada366f8d03919b3348ea71112e64","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.10@sha256:c3a18aebb8251339117ea998296315de17bada366f8d03919b3348ea71112e64"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.28.10","digest":"sha256:c06076f7aca95df713e0748c44d80c0a3c2538fad67bfdd04296d45158e083e6","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.28.10@sha256:c06076f7aca95df713e0748c44d80c0a3c2538fad67bfdd04296d45158e083e6"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.14","digest":"sha256:b2f0c2b2f17b5fbe809e5bb99dc185b6ddd70df25295dc63a6d526350334eff5","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.14@sha256:b2f0c2b2f17b5fbe809e5bb99dc185b6ddd70df25295dc63a6d526350334eff5"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:bac2192f6374d6262116399b34fc5e143d576f82719e90a18261cae7480f4d4e","pinned_image":"ghcr.io/github/gh-aw-node@sha256:bac2192f6374d6262116399b34fc5e143d576f82719e90a18261cae7480f4d4e"},{"image":"ghcr.io/github/github-mcp-server:v1.11.0","digest":"sha256:fbec75de11c255213fa08d80fb166abe73d851fff631c51c0079872967720699","pinned_image":"ghcr.io/github/github-mcp-server:v1.11.0@sha256:fbec75de11c255213fa08d80fb166abe73d851fff631c51c0079872967720699"}],"mcp_servers":[{"name":"github","tools":["actions_get","actions_list","get_code_scanning_alert","get_commit","get_discussion","get_discussion_comments","get_file_contents","get_job_logs","get_label","get_latest_release","get_me","get_notification_details","get_pull_request","get_pull_request_comments","get_pull_request_diff","get_pull_request_files","get_pull_request_review_comments","get_pull_request_reviews","get_pull_request_status","get_release_by_tag","get_secret_scanning_alert","get_tag","issue_read","list_branches","list_code_scanning_alerts","list_commits","list_discussion_categories","list_discussions","list_issue_types","list_issues","list_label","list_notifications","list_pull_requests","list_releases","list_secret_scanning_alerts","list_starred_repositories","list_tags","pull_request_read","search_code","search_issues","search_orgs","search_pull_requests","search_repositories","search_users"]},{"name":"safeoutputs","tools":["add_comment","add_labels","create_pull_request","missing_data","missing_tool","noop","push_to_pull_request_branch","remove_labels","update_issue"]}]} +# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"0e27ffba3e15ac2073376b6105a25754d338020cd3b9cb33c474a2518ce1e6b9","body_hash":"86eb8fb11893e97d78e6a56f65cd45ca7bde4e300ed2d9f810f268b0107e52bf","compiler_version":"v0.87.10","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.80"}} +# gh-aw-manifest: {"version":1,"secrets":["COPILOT_GITHUB_TOKEN","GH_AW_CI_TRIGGER_TOKEN","GH_AW_DEFAULT_OTLP_HEADERS","GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GITHUB_TOKEN"],"actions":[{"repo":"actions/cache/restore","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/save","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/checkout","sha":"11d5960a326750d5838078e36cf38b85af677262","version":"v4"},{"repo":"actions/checkout","sha":"3d3c42e5aac5ba805825da76410c181273ba90b1","version":"v7.0.1"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/setup-python","sha":"5fda3b95a4ea91299a34e894583c3862153e4b97","version":"v7.0.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"},{"repo":"github/gh-aw-actions/setup","sha":"bc8c008a419c5b7a29df6f5641edd35fd1c6ea85","version":"v0.87.10"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.28.10","digest":"sha256:c01e6d16d11ea4f2a46cc023a9f402224a3b3861b026818eec0dc586d7e6918e","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.28.10@sha256:c01e6d16d11ea4f2a46cc023a9f402224a3b3861b026818eec0dc586d7e6918e"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.10","digest":"sha256:c3a18aebb8251339117ea998296315de17bada366f8d03919b3348ea71112e64","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.10@sha256:c3a18aebb8251339117ea998296315de17bada366f8d03919b3348ea71112e64"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.28.10","digest":"sha256:c06076f7aca95df713e0748c44d80c0a3c2538fad67bfdd04296d45158e083e6","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.28.10@sha256:c06076f7aca95df713e0748c44d80c0a3c2538fad67bfdd04296d45158e083e6"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.14","digest":"sha256:b2f0c2b2f17b5fbe809e5bb99dc185b6ddd70df25295dc63a6d526350334eff5","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.14@sha256:b2f0c2b2f17b5fbe809e5bb99dc185b6ddd70df25295dc63a6d526350334eff5"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:bac2192f6374d6262116399b34fc5e143d576f82719e90a18261cae7480f4d4e","pinned_image":"ghcr.io/github/gh-aw-node@sha256:bac2192f6374d6262116399b34fc5e143d576f82719e90a18261cae7480f4d4e"},{"image":"ghcr.io/github/github-mcp-server:v1.11.0","digest":"sha256:fbec75de11c255213fa08d80fb166abe73d851fff631c51c0079872967720699","pinned_image":"ghcr.io/github/github-mcp-server:v1.11.0@sha256:fbec75de11c255213fa08d80fb166abe73d851fff631c51c0079872967720699"}],"mcp_servers":[{"name":"github","tools":["actions_get","actions_list","get_code_scanning_alert","get_commit","get_discussion","get_discussion_comments","get_file_contents","get_job_logs","get_label","get_latest_release","get_me","get_notification_details","get_pull_request","get_pull_request_comments","get_pull_request_diff","get_pull_request_files","get_pull_request_review_comments","get_pull_request_reviews","get_pull_request_status","get_release_by_tag","get_secret_scanning_alert","get_tag","issue_read","list_branches","list_code_scanning_alerts","list_commits","list_discussion_categories","list_discussions","list_issue_types","list_issues","list_label","list_notifications","list_pull_requests","list_releases","list_secret_scanning_alerts","list_starred_repositories","list_tags","pull_request_read","search_code","search_issues","search_orgs","search_pull_requests","search_repositories","search_users"]},{"name":"safeoutputs","tools":["add_comment","add_labels","create_pull_request","missing_data","missing_tool","noop","push_to_pull_request_branch","remove_labels","update_issue"]}]} # This file was automatically generated by gh-aw (v0.87.10). DO NOT EDIT. To debug this workflow, load the skill at https://github.com/github/gh-aw/blob/main/debug.md # # ___ _ _ @@ -44,6 +44,7 @@ # Custom actions used: # - actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 # - actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 +# - actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 # - actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 # - actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 # - actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 @@ -121,7 +122,7 @@ env: jobs: activation: needs: pre_activation - if: "needs.pre_activation.outputs.activated == 'true' && ((github.event_name == 'issues' || github.event_name == 'issue_comment' || github.event_name == 'pull_request' || github.event_name == 'pull_request_review_comment' || github.event_name == 'discussion' || github.event_name == 'discussion_comment') && (github.event_name == 'issues' && (startsWith(github.event.issue.body, '/goal ') || startsWith(github.event.issue.body, '/goal\n') || github.event.issue.body == '/goal') || github.event_name == 'issue_comment' && (startsWith(github.event.comment.body, '/goal ') || startsWith(github.event.comment.body, '/goal\n') || github.event.comment.body == '/goal') && github.event.issue.pull_request == null || github.event_name == 'issue_comment' && (startsWith(github.event.comment.body, '/goal ') || startsWith(github.event.comment.body, '/goal\n') || github.event.comment.body == '/goal') && github.event.issue.pull_request != null || github.event_name == 'pull_request_review_comment' && (startsWith(github.event.comment.body, '/goal ') || startsWith(github.event.comment.body, '/goal\n') || github.event.comment.body == '/goal') || github.event_name == 'pull_request' && (startsWith(github.event.pull_request.body, '/goal ') || startsWith(github.event.pull_request.body, '/goal\n') || github.event.pull_request.body == '/goal') || github.event_name == 'discussion' && (startsWith(github.event.discussion.body, '/goal ') || startsWith(github.event.discussion.body, '/goal\n') || github.event.discussion.body == '/goal') || github.event_name == 'discussion_comment' && (startsWith(github.event.comment.body, '/goal ') || startsWith(github.event.comment.body, '/goal\n') || github.event.comment.body == '/goal')) || !(github.event_name == 'issues' || github.event_name == 'issue_comment' || github.event_name == 'pull_request' || github.event_name == 'pull_request_review_comment' || github.event_name == 'discussion' || github.event_name == 'discussion_comment'))" + if: needs.pre_activation.outputs.activated == 'true' runs-on: ubuntu-slim permissions: actions: read @@ -413,7 +414,7 @@ jobs: GH_AW_INCLUDE_PR_CONTEXT: ${{ (github.event_name == 'issue_comment' && github.event.issue.pull_request != null) || github.event_name == 'pull_request_review_comment' || github.event_name == 'pull_request_review' }} GH_AW_MCP_CLI_SERVERS_LIST: "- `github` — run `github --help` to see available tools\n- `safeoutputs` — run `safeoutputs --help` to see available tools" GH_AW_MEMORY_BRANCH_NAME: 'memory/goal' - GH_AW_MEMORY_CONSTRAINTS: "\n\n**Constraints:**\n- **Allowed Files**: Only files matching patterns: *.md\n- **Max File Size**: 40960 bytes (0.04 MB) per file\n- **Max File Count**: 100 files per commit\n- **Max Patch Size**: 10240 bytes (10 KB) total per push (max: 1024 KB)\n" + GH_AW_MEMORY_CONSTRAINTS: "\n\n**Constraints:**\n- **Max File Size**: 40960 bytes (0.04 MB) per file\n- **Max File Count**: 100 files per commit\n- **Max Patch Size**: 10240 bytes (10 KB) total per push (max: 1024 KB)\n" GH_AW_MEMORY_DESCRIPTION: '' GH_AW_MEMORY_DIR: '/tmp/gh-aw/repo-memory/default/' GH_AW_MEMORY_TARGET_REPO: ' of the current repository' @@ -489,8 +490,10 @@ jobs: retention-days: 1 agent: - needs: activation - if: needs.activation.outputs.daily_ai_credits_exceeded != 'true' + needs: + - activation + - preflight + if: "((needs.preflight.outputs.should_run == 'true') && ((github.event_name == 'issues' || github.event_name == 'issue_comment' || github.event_name == 'pull_request' || github.event_name == 'pull_request_review_comment' || github.event_name == 'discussion' || github.event_name == 'discussion_comment') && (github.event_name == 'issues' && (startsWith(github.event.issue.body, '/goal ') || startsWith(github.event.issue.body, '/goal\n') || github.event.issue.body == '/goal') || github.event_name == 'issue_comment' && (startsWith(github.event.comment.body, '/goal ') || startsWith(github.event.comment.body, '/goal\n') || github.event.comment.body == '/goal') && github.event.issue.pull_request == null || github.event_name == 'issue_comment' && (startsWith(github.event.comment.body, '/goal ') || startsWith(github.event.comment.body, '/goal\n') || github.event.comment.body == '/goal') && github.event.issue.pull_request != null || github.event_name == 'pull_request_review_comment' && (startsWith(github.event.comment.body, '/goal ') || startsWith(github.event.comment.body, '/goal\n') || github.event.comment.body == '/goal') || github.event_name == 'pull_request' && (startsWith(github.event.pull_request.body, '/goal ') || startsWith(github.event.pull_request.body, '/goal\n') || github.event.pull_request.body == '/goal') || github.event_name == 'discussion' && (startsWith(github.event.discussion.body, '/goal ') || startsWith(github.event.discussion.body, '/goal\n') || github.event.discussion.body == '/goal') || github.event_name == 'discussion_comment' && (startsWith(github.event.comment.body, '/goal ') || startsWith(github.event.comment.body, '/goal\n') || github.event.comment.body == '/goal')) || !(github.event_name == 'issues' || github.event_name == 'issue_comment' || github.event_name == 'pull_request' || github.event_name == 'pull_request_review_comment' || github.event_name == 'discussion' || github.event_name == 'discussion_comment'))) && (needs.activation.outputs.daily_ai_credits_exceeded != 'true')" runs-on: ubuntu-latest permissions: read-all timeout-minutes: 60 @@ -1448,6 +1451,7 @@ jobs: - activation - agent - detection + - preflight - push_repo_memory - safe_outputs if: > @@ -2018,7 +2022,8 @@ jobs: bash "${RUNNER_TEMP}/gh-aw/actions/conclude_threat_detection.sh" /tmp/gh-aw/threat-detection/detection_result.json pre_activation: - if: "(github.event_name != 'issue_comment' && github.event_name != 'pull_request_review_comment' || contains(fromJSON('[\"OWNER\",\"MEMBER\",\"COLLABORATOR\"]'), github.event.comment.author_association)) && ((github.event_name == 'issues' || github.event_name == 'issue_comment' || github.event_name == 'pull_request' || github.event_name == 'pull_request_review_comment' || github.event_name == 'discussion' || github.event_name == 'discussion_comment') && (github.event_name == 'issues' && (startsWith(github.event.issue.body, '/goal ') || startsWith(github.event.issue.body, '/goal\n') || github.event.issue.body == '/goal') || github.event_name == 'issue_comment' && (startsWith(github.event.comment.body, '/goal ') || startsWith(github.event.comment.body, '/goal\n') || github.event.comment.body == '/goal') && github.event.issue.pull_request == null || github.event_name == 'issue_comment' && (startsWith(github.event.comment.body, '/goal ') || startsWith(github.event.comment.body, '/goal\n') || github.event.comment.body == '/goal') && github.event.issue.pull_request != null || github.event_name == 'pull_request_review_comment' && (startsWith(github.event.comment.body, '/goal ') || startsWith(github.event.comment.body, '/goal\n') || github.event.comment.body == '/goal') || github.event_name == 'pull_request' && (startsWith(github.event.pull_request.body, '/goal ') || startsWith(github.event.pull_request.body, '/goal\n') || github.event.pull_request.body == '/goal') || github.event_name == 'discussion' && (startsWith(github.event.discussion.body, '/goal ') || startsWith(github.event.discussion.body, '/goal\n') || github.event.discussion.body == '/goal') || github.event_name == 'discussion_comment' && (startsWith(github.event.comment.body, '/goal ') || startsWith(github.event.comment.body, '/goal\n') || github.event.comment.body == '/goal')) || !(github.event_name == 'issues' || github.event_name == 'issue_comment' || github.event_name == 'pull_request' || github.event_name == 'pull_request_review_comment' || github.event_name == 'discussion' || github.event_name == 'discussion_comment'))" + if: > + github.event_name != 'issue_comment' && github.event_name != 'pull_request_review_comment' || contains(fromJSON('["OWNER","MEMBER","COLLABORATOR"]'), github.event.comment.author_association) runs-on: ubuntu-slim env: GH_AW_RUNTIME_FEATURES: ${{ vars.GH_AW_RUNTIME_FEATURES }} @@ -2071,6 +2076,42 @@ jobs: const { main } = require(path.join(actionsDir, 'check_membership.cjs')); await main(); + preflight: + name: Check for Goal work without an agent + needs: activation + if: > + github.event_name == 'schedule' || github.event_name == 'workflow_dispatch' || (contains(fromJSON('["issue_comment","pull_request_review_comment","discussion_comment"]'), github.event_name) && + startsWith(github.event.comment.body, '/goal')) || (github.event_name == 'issues' && startsWith(github.event.issue.body, '/goal')) || + (github.event_name == 'pull_request' && startsWith(github.event.pull_request.body, '/goal')) || + (github.event_name == 'discussion' && + startsWith(github.event.discussion.body, '/goal')) + runs-on: ubuntu-latest + permissions: + contents: read + issues: read + outputs: + should_run: ${{ steps.evaluate.outputs.should_run }} + steps: + - name: Configure GH_HOST for enterprise compatibility + id: ghes-host-config + shell: bash + run: | # zizmor: ignore[github-env] - GITHUB_SERVER_URL is set by GitHub Actions, not user input. + # Derive GH_HOST from GITHUB_SERVER_URL so the gh CLI targets the correct + # GitHub instance (GHES/GHEC). On github.com this is a harmless no-op. + GH_HOST="${GITHUB_SERVER_URL#https://}" + GH_HOST="${GH_HOST#http://}" + echo "GH_HOST=${GH_HOST}" >> "$GITHUB_ENV" + - name: Check out trusted scheduling code + uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 + with: + persist-credentials: false + ref: ${{ github.event_name == 'workflow_dispatch' && github.sha || github.event.repository.default_branch }} + - name: Check active goals or explicit steering + id: evaluate + run: python3 .github/workflows/scripts/goal_preflight.py + env: + GITHUB_TOKEN: ${{ github.token }} + push_repo_memory: needs: - activation @@ -2139,8 +2180,7 @@ jobs: MAX_FILE_SIZE: 40960 MAX_FILE_COUNT: 100 MAX_PATCH_SIZE: 10240 - ALLOWED_EXTENSIONS: '[]' - FILE_GLOB_FILTER: "*.md" + ALLOWED_EXTENSIONS: '[".md"]' with: script: | const path = require('path'); diff --git a/.github/workflows/goal.md b/.github/workflows/goal.md index 1cf7300e3..bcaa5805f 100644 --- a/.github/workflows/goal.md +++ b/.github/workflows/goal.md @@ -17,6 +17,37 @@ on: permissions: read-all +jobs: + preflight: + name: Check for Goal work without an agent + # Cheap event filter; the helper checks exact commands and gh-aw still + # enforces command-author authorization before activating the agent. + if: >- + github.event_name == 'schedule' || github.event_name == 'workflow_dispatch' || + (contains(fromJSON('["issue_comment","pull_request_review_comment","discussion_comment"]'), github.event_name) && startsWith(github.event.comment.body, '/goal')) || + (github.event_name == 'issues' && startsWith(github.event.issue.body, '/goal')) || + (github.event_name == 'pull_request' && startsWith(github.event.pull_request.body, '/goal')) || + (github.event_name == 'discussion' && startsWith(github.event.discussion.body, '/goal')) + runs-on: ubuntu-latest + permissions: + contents: read + issues: read + outputs: + should_run: ${{ steps.evaluate.outputs.should_run }} + steps: + - name: Check out trusted scheduling code + uses: actions/checkout@v4 + with: + ref: ${{ github.event_name == 'workflow_dispatch' && github.sha || github.event.repository.default_branch }} + persist-credentials: false + - id: evaluate + name: Check active goals or explicit steering + env: + GITHUB_TOKEN: ${{ github.token }} + run: python3 .github/workflows/scripts/goal_preflight.py + +if: needs.preflight.outputs.should_run == 'true' + timeout-minutes: 60 max-daily-ai-credits: 200K @@ -74,7 +105,8 @@ tools: bash: true repo-memory: branch-name: memory/goal - file-glob: ["*.md"] + # Slashless globs in gh-aw v0.87.10 exclude files at the memory root. + allowed-extensions: [".md"] max-file-size: 40960 imports: @@ -98,6 +130,12 @@ engine: copilot You are the Goal workflow. Your job is to keep working an open GitHub issue labeled `goal` until its completion contract is satisfied by concrete evidence. +The deterministic preflight skips scheduled or untargeted runs when there are +no open, unfinished goal issues. Explicit issue requests and slash-command +steering still reach this workflow. API errors fail visibly rather than being +reported as an empty queue. The scheduler below still chooses the issue only +after durable repo-memory has been restored. + Take heed of slash-command instructions: "${{ steps.sanitized.outputs.text }}" If the slash-command text is non-empty, treat it as steering for the selected @@ -342,6 +380,11 @@ Summary: ## Completion +Verification must execute the implementation, not just materialize expected +snapshots. Count the assertions actually exercised and identify the runtime +used (for example, Wasm rather than its fallback). Missing-runtime early returns +and silently skipped tests are missing evidence, never a passing contract. + When the completion contract is satisfied: 1. Update the state file: `Status: completed`, `Completed: true`, and a diff --git a/.github/workflows/scripts/autoloop_policy.py b/.github/workflows/scripts/autoloop_policy.py new file mode 100644 index 000000000..7303c7b43 --- /dev/null +++ b/.github/workflows/scripts/autoloop_policy.py @@ -0,0 +1,118 @@ +"""Pure policy helpers for Autoloop's side-effecting scheduling entry point.""" +import re + + +STATE_FILE_MAX_BYTES = 30720 + + +def parse_machine_state(content): + """Read the machine-state table without interpreting research prose as state.""" + state = {} + section = re.search(r"## ⚙️ Machine State.*?\n(.*?)(?=\n## |\Z)", content, re.DOTALL) + if not section: + return state + for row in re.finditer(r"\|\s*(.+?)\s*\|\s*(.+?)\s*\|", section.group(0)): + raw_key, raw_val = (value.strip() for value in row.groups()) + if raw_key.lower() in ("field", "---", ":---", ":---:", "---:"): + continue + key = raw_key.lower().replace(" ", "_") + state[key] = None if raw_val in ("—", "-", "") else raw_val + for field in ("iteration_count", "consecutive_errors"): + if field in state: + try: + state[field] = int(state[field]) + except (ValueError, TypeError): + state[field] = 0 + for field in ("paused", "completed"): + if field in state: + state[field] = str(state[field]).lower() == "true" + state["recent_statuses"] = [ + value.strip().lower() for value in (state.get("recent_statuses") or "").split(",") + if value.strip() + ] + return state + + +def metric_direction(content): + """Return (direction, error); never reverse an explicit evaluation contract.""" + stripped = re.sub(r"^(\s*\s*\n)*", "", content, flags=re.DOTALL) + frontmatter = re.match(r"^---\s*\n(.*?)\n---\s*\n", stripped, re.DOTALL) + declared = None + if frontmatter: + matches = re.findall(r"^metric_direction:\s*(.*?)\s*$", frontmatter.group(1), re.MULTILINE) + if len(matches) > 1: + return None, "duplicate metric_direction fields" + if matches: + declared = matches[0].split("#", 1)[0].strip().strip("\"'") + if declared not in ("higher", "lower"): + return None, "invalid metric_direction: expected higher or lower" + evaluation = re.search(r"^## Evaluation\s*\n(.*?)(?=^## |\Z)", content, re.MULTILINE | re.DOTALL) + directions = set() + if evaluation: + prose = re.sub(r"[`*_]", "", evaluation.group(1)) + directions = set(re.findall(r"\b(higher|lower)\s+is\s+better\b", prose.lower())) + if len(directions) > 1 or (declared and directions and declared not in directions): + return None, "conflicting metric direction in the program contract" + if declared: + return declared, None + if directions: + return directions.pop(), None + return None, "metric direction missing from frontmatter and Evaluation" + + +def state_metadata(content): + """Count UTF-8 bytes, matching repo-memory's configured per-file bound.""" + return { + "state_file_size_bytes": len(content.encode("utf-8")), + "state_file_max_bytes": STATE_FILE_MAX_BYTES, + } + + +def _latest_entry(content, title): + section = re.search( + rf"^## [^\n]*\b{title}\b[^\n]*\n(.*?)(?=^## |\Z)", + content, re.MULTILINE | re.DOTALL, + ) + if not section: + return "" + body = section.group(1) + entry = re.search(r"^### [^\n]+\n.*?(?=^### |\Z)", body, re.MULTILINE | re.DOTALL) + if entry: + return entry.group(0) + # Older compact population files put the newest candidate on the first bullet. + bullet = re.search(r"^- [^\n]*\bc\d+\b[^\n]*$", body, re.MULTILINE) + return bullet.group(0) if bullet else "" + + +def pending_candidate_kind(state, content): + """Pending work is a reason to reconcile, never evidence for acceptance.""" + if state.get("pending_tree"): + return "tree" + if state.get("pending_run") or (state.get("recent_statuses") or [None])[-1] == "pending-ci": + return "legacy" + # Do not let an old entry or a prose mention revive a resolved candidate. + for title in ("Population", "Iteration History"): + entry = _latest_entry(content, title) + status = re.search(r"^\s*-?\s*\*{0,2}Status\*{0,2}:\s*(.+)$", entry, re.MULTILINE) + if status: + is_pending = re.search(r"\bpending[- ]ci\b", status.group(1), re.IGNORECASE) + elif entry.startswith("- "): + is_pending = re.search(r"\bpending[- ]ci\b", entry, re.IGNORECASE) + else: + is_pending = re.search(r"^\s*(?:⏳\s*)?Pending[- ]CI\b", entry, re.MULTILINE | re.IGNORECASE) + if is_pending: + return "legacy" + return None + + +def automatic_skip_reason(state, content): + """Explicit stops win; an automatic plateau cannot strand pending evidence.""" + if str(state.get("completed", "")).lower() == "true": + return "completed: target metric reached" + if str(state.get("paused", "")).lower() == "true": + return f"paused: {state.get('pause_reason') or 'unknown'}" + recent = state.get("recent_statuses", [])[-5:] + if len(recent) == 5 and all(status == "rejected" for status in recent): + if not pending_candidate_kind(state, content): + return "plateau: 5 consecutive rejections" + return None diff --git a/.github/workflows/scripts/autoloop_scheduler.py b/.github/workflows/scripts/autoloop_scheduler.py index f2ee5b8ee..157df593a 100644 --- a/.github/workflows/scripts/autoloop_scheduler.py +++ b/.github/workflows/scripts/autoloop_scheduler.py @@ -9,6 +9,10 @@ import os, json, re, glob, sys import urllib.request, urllib.error from datetime import datetime, timezone, timedelta +from autoloop_policy import ( + automatic_skip_reason, metric_direction, parse_machine_state, + pending_candidate_kind, state_metadata, +) programs_dir = ".autoloop/programs" autoloop_dir = ".autoloop/programs" @@ -73,49 +77,18 @@ def slugify(title): # The single repo-memory tool has id "default", independently of its branch name. # Use the same directory the framework clones and uploads after the agent. repo_memory_dir = "/tmp/gh-aw/repo-memory/default" - -def parse_machine_state(content): - """Parse the ⚙️ Machine State table from a state file. Returns a dict.""" - state = {} - m = re.search(r'## ⚙️ Machine State.*?\n(.*?)(?=\n## |\Z)', content, re.DOTALL) - if not m: - return state - section = m.group(0) - for row in re.finditer(r'\|\s*(.+?)\s*\|\s*(.+?)\s*\|', section): - raw_key = row.group(1).strip() - raw_val = row.group(2).strip() - if raw_key.lower() in ("field", "---", ":---", ":---:", "---:"): - continue - key = raw_key.lower().replace(" ", "_") - val = None if raw_val in ("—", "-", "") else raw_val - state[key] = val - # Coerce types - for int_field in ("iteration_count", "consecutive_errors"): - if int_field in state: - try: - state[int_field] = int(state[int_field]) - except (ValueError, TypeError): - state[int_field] = 0 - if "paused" in state: - state["paused"] = str(state.get("paused", "")).lower() == "true" - if "completed" in state: - state["completed"] = str(state.get("completed", "")).lower() == "true" - # recent_statuses: stored as comma-separated words (e.g. "accepted, rejected, error") - rs_raw = state.get("recent_statuses") or "" - if rs_raw: - state["recent_statuses"] = [s.strip().lower() for s in rs_raw.split(",") if s.strip()] - else: - state["recent_statuses"] = [] - return state +state_documents = {} def read_program_state(program_name): """Read scheduling state from the repo-memory state file.""" state_file = os.path.join(repo_memory_dir, f"{program_name}.md") if not os.path.isfile(state_file): + state_documents[program_name] = "" print(f" {program_name}: no state file found (first run)") return {} with open(state_file, encoding="utf-8") as f: content = f.read() + state_documents[program_name] = content return parse_machine_state(content) # Bootstrap: create autoloop programs directory and template if missing @@ -281,6 +254,8 @@ def read_program_state(program_name): skipped = [] unconfigured = [] all_programs = {} # name -> file path (populated during scanning) +program_directions = {} +program_reconciliation = {} # Schedule string to timedelta def parse_schedule(s): @@ -345,6 +320,8 @@ def get_program_name(pf): # Read state from repo-memory state = read_program_state(name) + program_directions[name] = metric_direction(content) + program_reconciliation[name] = pending_candidate_kind(state, state_documents[name]) if state: print(f" {name}: last_run={state.get('last_run')}, iteration_count={state.get('iteration_count')}") else: @@ -358,20 +335,11 @@ def get_program_name(pf): except ValueError: pass - # Check if completed (target metric was reached) - if str(state.get("completed", "")).lower() == "true": - skipped.append({"name": name, "reason": f"completed: target metric reached"}) - continue - - # Check if paused (e.g., plateau or recurring errors) - if state.get("paused"): - skipped.append({"name": name, "reason": f"paused: {state.get('pause_reason', 'unknown')}"}) - continue - - # Auto-pause on plateau: 5+ consecutive rejections - recent = state.get("recent_statuses", [])[-5:] - if len(recent) >= 5 and all(s == "rejected" for s in recent): - skipped.append({"name": name, "reason": "plateau: 5 consecutive rejections"}) + # An explicit stop wins; automatic plateau detection must not strand + # an unresolved current or legacy candidate before the agent can reconcile it. + stop_reason = automatic_skip_reason(state, state_documents[name]) + if stop_reason: + skipped.append({"name": name, "reason": stop_reason}) continue # Check if due based on per-program schedule @@ -544,6 +512,10 @@ def verify_pr_is_open(pr_number): "selected_file": selected_file, "selected_issue": selected_issue, "selected_target_metric": selected_target_metric, + "selected_metric_direction": program_directions.get(selected, (None, None))[0], + "selected_metric_direction_error": program_directions.get(selected, (None, None))[1], + "selected_reconciliation": program_reconciliation.get(selected), + **state_metadata(state_documents.get(selected, "")), "existing_pr": existing_pr, "head_branch": head_branch, "issue_programs": {name: info["issue_number"] for name, info in issue_programs.items()}, diff --git a/.github/workflows/scripts/fixtures/gh-aw-v0.87.10/LICENSE b/.github/workflows/scripts/fixtures/gh-aw-v0.87.10/LICENSE new file mode 100644 index 000000000..28a50fa22 --- /dev/null +++ b/.github/workflows/scripts/fixtures/gh-aw-v0.87.10/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright GitHub, Inc. + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/.github/workflows/scripts/fixtures/gh-aw-v0.87.10/SOURCE.md b/.github/workflows/scripts/fixtures/gh-aw-v0.87.10/SOURCE.md new file mode 100644 index 000000000..008346879 --- /dev/null +++ b/.github/workflows/scripts/fixtures/gh-aw-v0.87.10/SOURCE.md @@ -0,0 +1,20 @@ +# Pinned memory transport fixtures + +These are unmodified functions from +[`github/gh-aw-actions` at `bc8c008a419c5b7a29df6f5641edd35fd1c6ea85`](https://github.com/github/gh-aw-actions/tree/bc8c008a419c5b7a29df6f5641edd35fd1c6ea85/setup/js), +the runtime pinned by the generated v0.87.10 workflows: + +- `glob_pattern_helpers.cjs`: Git blob `7f8f41db39abc0d1fdecc5a04fda79123fd7f9e1`. +- `validate_memory_files.cjs`: Git blob `2fe8bf15b7611c029ce5631e27b69f9a70a96dd0`. + +`push_repo_memory.cjs` (blob `b1297b11f3860fc6275c3ffd269ea2f7877b3f9b`, lines +323–328) invokes the matcher with `matchSubfolderRoot: !pattern.includes("/")`. +Consequently, slashless filters exclude root files, while slash-containing +filters require a slash. No supported file-glob can select root Markdown in +this version. We use the explicit `.md` extension allowlist and no glob instead. + +The regression test executes both real helpers, preserves their Git blob +checksums, and exercises root/nested Markdown acceptance and non-Markdown +rejection. It stubs only the validator's error-message helper and log sink; +it never pushes commits or calls GitHub. Review this fixture when upgrading +the workflow runtime. See `LICENSE` for the upstream MIT license. diff --git a/.github/workflows/scripts/fixtures/gh-aw-v0.87.10/glob_pattern_helpers.cjs b/.github/workflows/scripts/fixtures/gh-aw-v0.87.10/glob_pattern_helpers.cjs new file mode 100644 index 000000000..7f8f41db3 --- /dev/null +++ b/.github/workflows/scripts/fixtures/gh-aw-v0.87.10/glob_pattern_helpers.cjs @@ -0,0 +1,171 @@ +// @ts-check + +/** + * Internal helper to escape special regex characters in a pattern + * @param {string} pattern - Pattern to escape + * @returns {string} - Pattern with special characters escaped + * @private + */ +function escapeRegexChars(pattern) { + // Escape backslashes first, then dots, then other special chars + return pattern + .replace(/\\/g, "\\\\") // Escape backslashes + .replace(/\./g, "\\.") // Escape dots + .replace(/[+?^${}()|[\]]/g, "\\$&"); // Escape other special regex chars (except * which is handled separately) +} + +/** + * Convert a glob pattern to a RegExp + * @param {string} pattern - Glob pattern (e.g., "*.json", "metrics/**", "data/**\/*.csv") + * @param {Object} [options] - Options for pattern conversion + * @param {boolean} [options.pathMode=true] - If true, * matches non-slash chars; if false, * matches any char + * @param {boolean} [options.caseSensitive=true] - Whether matching should be case-sensitive + * @param {boolean} [options.matchSubfolderRoot=false] - If true and the pattern contains no "/", + * the pattern is required to be prefixed by exactly one path segment (i.e. matches only at the + * root of a single memory subfolder: "subfolder/file.json"). Files at the artifact root (depth 0) + * and files deeper than one level (depth 2+) are not matched. + * @returns {RegExp} - Regular expression that matches the pattern + * + * Supports: + * - * matches any characters except / (in path mode) or any characters (in simple mode) + * - ** matches any characters including / (only in path mode) + * - . is escaped to match literal dots + * - \ is escaped properly + * + * @example + * const regex = globPatternToRegex("*.json"); + * regex.test("file.json"); // true + * regex.test("file.txt"); // false + * + * @example + * const regex = globPatternToRegex("metrics/**"); + * regex.test("metrics/data.json"); // true + * regex.test("metrics/daily/data.json"); // true + * + * @example + * const regex = globPatternToRegex("*.json", { matchSubfolderRoot: true }); + * regex.test("discussion-task-miner/processed-discussions.json"); // true (depth 1) + * regex.test("file.json"); // false (depth 0 – not matched) + * regex.test("discussion-task-miner/archive/old.json"); // false (depth 2 – not matched) + */ +function globPatternToRegex(pattern, options) { + const { pathMode = true, caseSensitive = true, matchSubfolderRoot = false } = options || {}; + + let regexPattern = escapeRegexChars(pattern); + + if (pathMode) { + // Path mode: handle ** and * differently + regexPattern = regexPattern + .replace(/\*\*/g, "") // Temporarily replace ** + .replace(/\*/g, "[^/]*") // Single * matches non-slash chars + .replace(//g, ".*"); // ** matches everything including / + } else { + // Simple mode: * matches any character + regexPattern = regexPattern.replace(/\*/g, ".*"); + } + + // matchSubfolderRoot: slashless patterns must match exactly at the root of one subfolder + // (depth 1 only – neither depth 0 nor depth 2+). + // Patterns that already contain a "/" are matched against the full relative path unchanged. + if (matchSubfolderRoot && !pattern.includes("/")) { + regexPattern = `[^/]+/${regexPattern}`; + } + + // eslint-disable-next-line gh-aw-custom/require-escaped-regexp-interpolation -- regexPattern is intentionally built as a regex: * and ** are replaced with regex patterns after escaping all other metacharacters + return new RegExp(`^${regexPattern}$`, caseSensitive ? "" : "i"); +} + +/** + * Parse a space-separated list of glob patterns into RegExp objects + * @param {string} fileGlobFilter - Space-separated glob patterns (e.g., "*.json *.jsonl *.csv *.md") + * @returns {RegExp[]} - Array of regular expressions + * + * @example + * const patterns = parseGlobPatterns("*.json *.jsonl"); + * patterns[0].test("file.json"); // true + * patterns[1].test("file.jsonl"); // true + */ +function parseGlobPatterns(fileGlobFilter) { + return fileGlobFilter + .trim() + .split(/\s+/) + .filter(Boolean) + .map(pattern => globPatternToRegex(pattern)); +} + +/** + * Check if a file path matches any of the provided glob patterns + * @param {string} filePath - File path to test (e.g., "data/file.json") + * @param {string} fileGlobFilter - Space-separated glob patterns + * @returns {boolean} - True if the file matches at least one pattern + * + * @example + * matchesGlobPattern("file.json", "*.json *.jsonl"); // true + * matchesGlobPattern("file.txt", "*.json *.jsonl"); // false + */ +function matchesGlobPattern(filePath, fileGlobFilter) { + const patterns = parseGlobPatterns(fileGlobFilter); + return patterns.some(pattern => pattern.test(filePath)); +} + +/** + * Convert a simple glob pattern to a RegExp (for non-path matching) + * @param {string} pattern - Glob pattern (e.g., "copilot", "*[bot]") + * @param {boolean} caseSensitive - Whether matching should be case-sensitive (default: false) + * @returns {RegExp} - Regular expression that matches the pattern + * + * Supports: + * - * matches any characters (not limited to non-slash like path mode) + * - Escapes special regex characters except * + * - Case-insensitive by default + * + * @example + * const regex = simpleGlobToRegex("*[bot]"); + * regex.test("dependabot[bot]"); // true + * regex.test("github-actions[bot]"); // true + * + * @example + * const regex = simpleGlobToRegex("copilot"); + * regex.test("copilot"); // true + * regex.test("Copilot"); // true (case-insensitive) + */ +function simpleGlobToRegex(pattern, caseSensitive = false) { + return globPatternToRegex(pattern, { pathMode: false, caseSensitive }); +} + +/** + * Check if a string matches a simple glob pattern + * @param {string} str - String to test (e.g., "copilot", "dependabot[bot]") + * @param {string} pattern - Glob pattern (e.g., "copilot", "*[bot]") + * @param {boolean} caseSensitive - Whether matching should be case-sensitive (default: false) + * @returns {boolean} - True if the string matches the pattern + * + * @example + * matchesSimpleGlob("dependabot[bot]", "*[bot]"); // true + * matchesSimpleGlob("copilot", "copilot"); // true + * matchesSimpleGlob("Copilot", "copilot"); // true (case-insensitive by default) + * matchesSimpleGlob("alice", "*[bot]"); // false + */ +function matchesSimpleGlob(str, pattern, caseSensitive = false) { + if (!str || !pattern) { + return false; + } + + // Exact match check (case-insensitive by default) + if (!caseSensitive && str.toLowerCase() === pattern.toLowerCase()) { + return true; + } else if (caseSensitive && str === pattern) { + return true; + } + + const regex = simpleGlobToRegex(pattern, caseSensitive); + return regex.test(str); +} + +module.exports = { + globPatternToRegex, + parseGlobPatterns, + matchesGlobPattern, + simpleGlobToRegex, + matchesSimpleGlob, +}; diff --git a/.github/workflows/scripts/fixtures/gh-aw-v0.87.10/validate_memory_files.cjs b/.github/workflows/scripts/fixtures/gh-aw-v0.87.10/validate_memory_files.cjs new file mode 100644 index 000000000..2fe8bf15b --- /dev/null +++ b/.github/workflows/scripts/fixtures/gh-aw-v0.87.10/validate_memory_files.cjs @@ -0,0 +1,89 @@ +// @ts-check +/// + +const fs = require("fs"); +const path = require("path"); +const { getErrorMessage } = require("./error_helpers.cjs"); + +/** + * @typedef {Object} ValidationResult + * @property {boolean} valid - Whether all files passed validation + * @property {string[]} invalidFiles - List of files with invalid extensions + */ + +/** + * Validate that all files in a memory directory have allowed file extensions + * If allowedExtensions is empty or not provided, all file extensions are allowed + * + * @param {string} memoryDir - Path to the memory directory to validate + * @param {string} [memoryType="cache"] - Type of memory ("cache" or "repo") for error messages + * @param {string[]} [allowedExtensions] - Optional custom list of allowed extensions (empty array or undefined means allow all files) + * @param {{ info: (message: string) => void, error: (message: string) => void }} [coreModule] - Actions core module + * @returns {ValidationResult} Validation result with list of invalid files + */ +function validateMemoryFiles(memoryDir, memoryType = "cache", allowedExtensions, coreModule = core) { + if (!allowedExtensions?.length) { + coreModule.info(`All file extensions are allowed in ${memoryType}-memory directory`); + return { valid: true, invalidFiles: [] }; + } + + if (!fs.existsSync(memoryDir)) { + coreModule.info(`Memory directory does not exist: ${memoryDir}`); + return { valid: true, invalidFiles: [] }; + } + + const extensions = new Set(allowedExtensions.map(ext => ext.trim().toLowerCase())); + /** @type {string[]} */ + const invalidFiles = []; + + /** + * Recursively scan directory for files + * @param {string} dirPath - Directory to scan + * @param {string} [relativePath=""] - Relative path from memory directory + */ + const scanDirectory = (dirPath, relativePath = "") => { + const entries = fs.readdirSync(dirPath, { withFileTypes: true }); + + for (const entry of entries) { + const fullPath = path.join(dirPath, entry.name); + const relativeFilePath = relativePath ? path.join(relativePath, entry.name) : entry.name; + + if (entry.isDirectory()) { + // Skip .git directory — it is git metadata used for integrity branching + // and contains files with no extension (e.g. HEAD, ORIG_HEAD, packed-refs). + if (entry.name === ".git") continue; + scanDirectory(fullPath, relativeFilePath); + } else if (entry.isFile()) { + const ext = path.extname(entry.name).toLowerCase(); + if (!extensions.has(ext)) { + invalidFiles.push(relativeFilePath); + } + } + } + }; + + try { + scanDirectory(memoryDir); + } catch (error) { + const message = getErrorMessage(error); + coreModule.error(`Failed to scan ${memoryType}-memory directory: ${message}`); + return { valid: false, invalidFiles: [] }; + } + + if (invalidFiles.length > 0) { + coreModule.error(`Found ${invalidFiles.length} file(s) with invalid extensions in ${memoryType}-memory:`); + for (const file of invalidFiles) { + const ext = path.extname(file).toLowerCase() || "(no extension)"; + coreModule.error(` - ${file} (extension: ${ext})`); + } + coreModule.error(`Allowed extensions: ${[...extensions].join(", ")}`); + return { valid: false, invalidFiles }; + } + + coreModule.info(`All files in ${memoryType}-memory directory have valid extensions`); + return { valid: true, invalidFiles: [] }; +} + +module.exports = { + validateMemoryFiles, +}; diff --git a/.github/workflows/scripts/goal_preflight.py b/.github/workflows/scripts/goal_preflight.py new file mode 100644 index 000000000..b84d05f9e --- /dev/null +++ b/.github/workflows/scripts/goal_preflight.py @@ -0,0 +1,134 @@ +"""Skip idle Goal runs before gh-aw initializes a model or its tools. + +This only discovers whether work exists. The normal scheduler retains ownership +of selection and reads restored repo-memory; explicit steering must not be lost +just because it names an issue that is not yet labeled as a goal. +""" +from __future__ import annotations + +import json +import os +import re +import sys +import urllib.error +import urllib.parse +import urllib.request + + +class PreflightError(RuntimeError): + """Discovery failed; an empty queue has not been established.""" + + +def _get_page(url: str, token: str): + request = urllib.request.Request(url, headers={ + "Authorization": "Bearer " + token, + "Accept": "application/vnd.github+json", + }) + try: + with urllib.request.urlopen(request, timeout=30) as response: + return json.load(response), response.headers.get("Link", "") + except (urllib.error.URLError, ValueError, OSError) as error: + # Do not include response bodies or credential-bearing request objects. + raise PreflightError("Could not read goal issues from GitHub") from error + + +def _next_page(link_header: str, initial_url: str) -> str | None: + for part in link_header.split(","): + match = re.match(r'^\s*<([^>]+)>;\s*rel="next"\s*$', part) + if match: + url = match.group(1) + candidate = urllib.parse.urlsplit(url) + initial = urllib.parse.urlsplit(initial_url) + if ((candidate.scheme, candidate.netloc, candidate.path) != + (initial.scheme, initial.netloc, initial.path)): + raise PreflightError("Unexpected pagination URL from GitHub") + return url + return None + + +def has_open_goal(repo: str, token: str, api_url: str = "https://api.github.com", + get_page=_get_page) -> bool: + if not token or not re.fullmatch(r"[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+", repo): + raise PreflightError("Goal discovery requires a repository and GitHub token") + initial_url = (api_url.rstrip("/") + "/repos/" + repo + + "/issues?labels=goal&state=open&per_page=100") + url = initial_url + visited = set() + while url: + if url in visited: + raise PreflightError("GitHub repeated a goal issue page") + visited.add(url) + body, link_header = get_page(url, token) + if not isinstance(body, list): + raise PreflightError("GitHub did not return a goal issue list") + for issue in body: + if not isinstance(issue, dict): + raise PreflightError("GitHub returned an invalid goal issue") + if issue.get("pull_request") or issue.get("state") != "open": + continue + labels = issue.get("labels") + if not isinstance(labels, list) or any( + not isinstance(label, dict) or not isinstance(label.get("name"), str) + for label in labels + ): + raise PreflightError("GitHub returned invalid goal labels") + names = {label["name"] for label in labels} + if "goal" in names and "goal-completed" not in names: + return True + url = _next_page(link_header, initial_url) + return False + + +def has_goal_command(event_name: str, event: dict) -> bool: + if event_name in ("issue_comment", "pull_request_review_comment", "discussion_comment"): + entity = event.get("comment", {}) + elif event_name in ("issues", "pull_request", "discussion"): + entity = event.get({"issues": "issue"}.get(event_name, event_name), {}) + else: + return False + body = entity.get("body", "") if isinstance(entity, dict) else "" + return isinstance(body, str) and ( + body == "/goal" or body.startswith("/goal ") or body.startswith("/goal\n") + ) + + +def should_run(event_name: str, event: dict, discover) -> tuple[bool, str]: + if has_goal_command(event_name, event): + # gh-aw's existing author/command authorization still applies. + return True, "Explicit goal steering" + if event_name == "workflow_dispatch": + issue = (event.get("inputs") or {}).get("issue", "") + if str(issue or "").strip(): + return True, "Explicit goal issue request" + elif event_name != "schedule": + return False, "No goal command" + if discover(): + return True, "Open unfinished goals exist" + return False, "No open unfinished goal issues; no model needed" + + +def main() -> int: + try: + with open(os.environ["GITHUB_EVENT_PATH"], encoding="utf-8") as handle: + event = json.load(handle) + if not isinstance(event, dict): + raise PreflightError("Invalid GitHub event") + decision, reason = should_run( + os.environ.get("GITHUB_EVENT_NAME", ""), event, + lambda: has_open_goal( + os.environ.get("GITHUB_REPOSITORY", ""), + os.environ.get("GITHUB_TOKEN", ""), + os.environ.get("GITHUB_API_URL", "https://api.github.com"), + ), + ) + with open(os.environ["GITHUB_OUTPUT"], "a", encoding="utf-8") as handle: + handle.write("should_run=" + str(decision).lower() + "\n") + print(reason) + return 0 + except (PreflightError, OSError, ValueError, KeyError) as error: + print("::error::Goal preflight failed: " + str(error), file=sys.stderr) + return 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/.github/workflows/scripts/goal_scheduler.py b/.github/workflows/scripts/goal_scheduler.py index bdeea52ae..3957f00c1 100644 --- a/.github/workflows/scripts/goal_scheduler.py +++ b/.github/workflows/scripts/goal_scheduler.py @@ -29,6 +29,11 @@ OUTPUT_DIR = "/tmp/gh-aw" OUTPUT_FILE = os.path.join(OUTPUT_DIR, "goal.json") + +class SchedulerError(RuntimeError): + """Required GitHub evidence is unavailable; do not schedule a guessed state.""" + + REQUIRED_SECTIONS = { "goal": ("goal",), "completion_contract": ("completion contract", "definition of done"), @@ -86,8 +91,8 @@ def _http_get_json(url: str, headers: dict[str, str], timeout: int = 30): body = json.loads(response.read().decode()) link_header = response.headers.get("link") or response.headers.get("Link") return body, link_header - except (urllib.error.URLError, urllib.error.HTTPError, ValueError, OSError): - return None, None + except (urllib.error.URLError, ValueError, OSError) as error: + raise SchedulerError("Could not read required scheduling evidence from GitHub") from error def extract_markdown_sections(markdown: str) -> dict[str, str]: @@ -200,7 +205,7 @@ def fetch_goal_issues(repo: str, github_token: str, http_get_json=_http_get_json """Fetch open issues with the goal label.""" if not repo or not github_token: - return [] + raise SchedulerError("Goal discovery requires a repository and GitHub token") headers = { "Authorization": "token {}".format(github_token), @@ -212,14 +217,26 @@ def fetch_goal_issues(repo: str, github_token: str, http_get_json=_http_get_json "?labels={}&state=open&per_page=100".format(repo, label) ) issues = [] + visited = set() while next_url: + if next_url in visited: + raise SchedulerError("GitHub repeated a goal issue page") + visited.add(next_url) body, link_header = http_get_json(next_url, headers) if not isinstance(body, list): - break + raise SchedulerError("GitHub did not return a goal issue list") for issue in body: - if not isinstance(issue, dict) or issue.get("pull_request"): + if not isinstance(issue, dict): + raise SchedulerError("GitHub returned an invalid goal issue") + if issue.get("pull_request"): continue - labels = [label_obj.get("name") for label_obj in issue.get("labels", [])] + label_objects = issue.get("labels") + if not isinstance(label_objects, list) or any( + not isinstance(label_obj, dict) or not isinstance(label_obj.get("name"), str) + for label_obj in label_objects + ): + raise SchedulerError("GitHub returned invalid goal labels") + labels = [label_obj["name"] for label_obj in label_objects] if COMPLETED_LABEL in labels: continue issues.append(issue) @@ -231,7 +248,7 @@ def find_existing_pr_for_branch(repo: str, branch: str, github_token: str, http_ """Return the open PR number for a branch, if one exists.""" if not repo or not branch or not github_token: - return None + raise SchedulerError("PR discovery requires a repository, branch, and GitHub token") owner = repo.split("/", 1)[0] headers = { "Authorization": "token {}".format(github_token), @@ -240,10 +257,15 @@ def find_existing_pr_for_branch(repo: str, branch: str, github_token: str, http_ head = urllib.parse.quote("{}:{}".format(owner, branch), safe="") url = "https://api.github.com/repos/{}/pulls?head={}&state=open".format(repo, head) body, _ = http_get_json(url, headers) - if isinstance(body, list) and body: + if not isinstance(body, list): + raise SchedulerError("GitHub did not return a pull request list") + if body: + if not isinstance(body[0], dict): + raise SchedulerError("GitHub returned an invalid pull request") number = body[0].get("number") - if number: - return number + if not isinstance(number, int) or isinstance(number, bool) or number <= 0: + raise SchedulerError("GitHub returned an invalid pull request number") + return number return None @@ -301,13 +323,16 @@ def main() -> int: os.makedirs(OUTPUT_DIR, exist_ok=True) - issues = fetch_goal_issues(repo, github_token) - goals = [issue_to_goal(issue, repo, github_token) for issue in issues] - selected, deferred, error = select_goal(goals, forced_issue) + try: + issues = fetch_goal_issues(repo, github_token) + goals = [issue_to_goal(issue, repo, github_token) for issue in issues] + selected, deferred, error = select_goal(goals, forced_issue) + except SchedulerError as failure: + goals, selected, deferred, error = [], None, [], str(failure) output = { "generated_at": datetime.now(timezone.utc).isoformat(), - "no_goals": not goals, + "no_goals": not goals if not error else False, "selected": selected, "deferred": deferred, "error": error, diff --git a/.github/workflows/scripts/test_autoloop_evidence_contract.py b/.github/workflows/scripts/test_autoloop_evidence_contract.py new file mode 100644 index 000000000..613694ce2 --- /dev/null +++ b/.github/workflows/scripts/test_autoloop_evidence_contract.py @@ -0,0 +1,64 @@ +"""Source-contract checks for guardrails; live runs still verify agent behavior.""" +from pathlib import Path +import unittest + + +WORKFLOWS = Path(__file__).resolve().parents[1] + + +class AutoloopEvidenceContractTest(unittest.TestCase): + @classmethod + def setUpClass(cls): + cls.source = (WORKFLOWS / "autoloop.md").read_text() + cls.guard = cls.source.split("## Objective And Evidence Guard\n", 1)[1].split( + "## Command Mode\n", 1 + )[0] + cls.direction = cls.source.split("### Metric Direction\n", 1)[1].split( + "## Program Definition\n", 1 + )[0] + + def test_stale_memory_cannot_expand_scope_or_pad_metrics(self): + for requirement in ( + "every mode and program-specific strategy", + "repo-memory is working history, not permission", + "bulk domain generation", + "one small, useful checkpoint", + "metric-contract-mismatch", + "edit protected program definitions or issue #1", + "reset instead", + ): + with self.subTest(requirement=requirement): + self.assertIn(requirement, self.guard) + + def test_parity_requires_independent_executed_behavior(self): + for requirement in ( + "Tests must execute the tsb operation", + "independently generated pandas results", + "not\n differential evidence", + "in-scope denominator and a verified numerator", + "completeness as\n unknown", + ): + with self.subTest(requirement=requirement): + self.assertIn(requirement, self.guard) + + def test_performance_requires_equivalent_measured_work(self): + for requirement in ( + "verify equivalent outputs", + "measured SHA, repeated\n timings, and variability", + "Separate cold work from repeated-input cache hits", + "Do not add import-time JIT primers", + "counts or successful parsing do not prove execution or speedup", + ): + with self.subTest(requirement=requirement): + self.assertIn(requirement, self.guard) + + def test_missing_scheduler_direction_uses_contract_or_stops(self): + self.assertIn("If the scheduler field is missing", self.direction) + self.assertIn("program frontmatter or Evaluation prose", self.direction) + self.assertIn("ambiguous or\nconflicting, stop", self.direction) + self.assertNotIn("defaults to `higher` when unset", self.source) + self.assertNotIn("program falls back to `higher`", self.source) + + +if __name__ == "__main__": + unittest.main() diff --git a/.github/workflows/scripts/test_autoloop_policy.py b/.github/workflows/scripts/test_autoloop_policy.py new file mode 100644 index 000000000..558b90a1d --- /dev/null +++ b/.github/workflows/scripts/test_autoloop_policy.py @@ -0,0 +1,120 @@ +"""Executable policy regressions without running the scheduler's API/file writes.""" +from pathlib import Path +import re +import unittest + +from autoloop_policy import ( + automatic_skip_reason, metric_direction, parse_machine_state, + pending_candidate_kind, state_metadata, +) + + +WORKFLOWS = Path(__file__).resolve().parents[1] + + +class AutoloopPolicyTest(unittest.TestCase): + def test_reads_current_machine_state_fields(self): + state = parse_machine_state("""## ⚙️ Machine State +| Field | Value | +|-------|-------| +| Pending Tree | abc123 | +| Pending Run | https://example/run | +| Paused | false | +| Completed | true | +| Iteration Count | 42 | +| Recent Statuses | accepted, pending-ci | +## Lessons +| Paused | true | +""") + self.assertEqual(state["pending_tree"], "abc123") + self.assertEqual(state["iteration_count"], 42) + self.assertEqual(state["recent_statuses"], ["accepted", "pending-ci"]) + self.assertFalse(state["paused"]) + self.assertTrue(state["completed"]) + + def test_direction_from_frontmatter_or_explicit_evaluation_prose(self): + self.assertEqual(metric_direction("---\nmetric_direction: lower # ratio\n---\n"), ("lower", None)) + for value in ("higher", "lower"): + self.assertEqual(metric_direction(f"## Evaluation\n**{value.title()} is better.**\n"), (value, None)) + + def test_ambiguous_invalid_or_conflicting_direction_is_not_defaulted(self): + for content in ( + "## Evaluation\nA score.\n", + "---\nmetric_direction: backwards\n---\n", + "---\nmetric_direction: lower\nmetric_direction: higher\n---\n", + "---\nmetric_direction: higher\n---\n## Evaluation\nLower is better.\n", + "## Evaluation\nLower is better. Higher is better.\n", + "## Goal\nLower is better.\n## Evaluation\nA score.\n", + ): + with self.subTest(content=content): + direction, error = metric_direction(content) + self.assertIsNone(direction) + self.assertIsNotNone(error) + + def test_repository_program_contracts_resolve_without_definition_edits(self): + programs = WORKFLOWS.parents[1] / ".autoloop" / "programs" + for name, expected in (("perf-comparison", "higher"), ("tsb-perf-evolve", "lower")): + self.assertEqual(metric_direction((programs / name / "program.md").read_text()), (expected, None)) + + def test_current_pending_tree_or_run_is_reconciled_before_plateau(self): + for field, expected in (("pending_tree", "tree"), ("pending_run", "legacy")): + state = {field: "published-evidence", "recent_statuses": ["rejected"] * 5} + self.assertEqual(pending_candidate_kind(state, ""), expected) + self.assertIsNone(automatic_skip_reason(state, "")) + + def test_latest_legacy_population_and_history_are_reconciled_before_plateau(self): + state = {"recent_statuses": ["pending-ci"] + ["rejected"] * 9} + for content in ( + "## 🧬 Population (summary)\n- **c086** (gen 86, ⏳ pending-ci): commit abc.\n- **c085** rejected.\n", + "## 📊 Iteration History\n### Iteration 86\n⏳ Pending CI | commit abc.\n### Iteration 85\nRejected.\n", + ): + self.assertEqual(pending_candidate_kind(state, content), "legacy") + self.assertIsNone(automatic_skip_reason(state, content)) + + def test_old_pending_or_future_prose_does_not_revive_resolved_work(self): + content = """## 🧬 Population (summary) +- **c087** accepted. +- **c086** pending-ci. +## 📊 Iteration History +### Iteration 87 +Accepted. +### Iteration 86 +Pending CI. +## Future Directions +Resolve pending-ci someday. +""" + state = {"recent_statuses": ["pending-ci"] + ["rejected"] * 9} + self.assertIsNone(pending_candidate_kind(state, content)) + self.assertEqual(automatic_skip_reason(state, content), "plateau: 5 consecutive rejections") + + def test_explicit_pause_and_completion_win_over_pending_work(self): + for field in ("paused", "completed"): + for value in (True, "true"): + state = {field: value, "pending_tree": "abc", "pause_reason": "human decision"} + self.assertTrue(automatic_skip_reason(state, "").startswith(field)) + + def test_resolved_entry_notes_cannot_revive_pending_work(self): + content = """## 🧬 Population +### Candidate c087 +- **Status**: accepted +- **Notes**: Reconciled the previous pending-ci attempt. +## 📊 Iteration History +### Iteration 87 +Accepted; previous pending-ci is resolved. +""" + self.assertIsNone(pending_candidate_kind({}, content)) + + def test_latest_current_pending_status_can_reconcile_without_tree(self): + state = {"recent_statuses": ["rejected"] * 5 + ["pending-ci"]} + self.assertEqual(pending_candidate_kind(state, ""), "legacy") + + def test_state_size_matches_utf8_and_declared_memory_limit(self): + metadata = state_metadata("🧬 café") + self.assertEqual(metadata["state_file_size_bytes"], len("🧬 café".encode("utf-8"))) + declared = re.search(r"^ max-file-size: (\d+)$", (WORKFLOWS / "autoloop.md").read_text(), re.MULTILINE) + self.assertEqual(metadata["state_file_max_bytes"], int(declared.group(1))) + self.assertEqual(state_metadata("")["state_file_size_bytes"], 0) + + +if __name__ == "__main__": + unittest.main() diff --git a/.github/workflows/scripts/test_goal_evidence_contract.py b/.github/workflows/scripts/test_goal_evidence_contract.py new file mode 100644 index 000000000..7b1b6518b --- /dev/null +++ b/.github/workflows/scripts/test_goal_evidence_contract.py @@ -0,0 +1,21 @@ +"""Prevent evidence-shaped no-ops from satisfying runtime-dependent goals.""" +from pathlib import Path +import unittest + + +class GoalEvidenceContractTest(unittest.TestCase): + def test_completion_requires_executed_assertions_and_runtime_identity(self): + source = (Path(__file__).resolve().parents[1] / "goal.md").read_text() + completion = source.split("## Completion\n", 1)[1].split("## Blocked Runs\n", 1)[0] + for requirement in ( + "must execute the implementation", "not just materialize expected", + "Count the assertions actually exercised", "identify the runtime", + "Wasm rather than its fallback", "Missing-runtime early returns", + "silently skipped tests are missing evidence", + ): + with self.subTest(requirement=requirement): + self.assertIn(requirement, completion) + + +if __name__ == "__main__": + unittest.main() diff --git a/.github/workflows/scripts/test_goal_preflight.py b/.github/workflows/scripts/test_goal_preflight.py new file mode 100644 index 000000000..10b7a4be2 --- /dev/null +++ b/.github/workflows/scripts/test_goal_preflight.py @@ -0,0 +1,145 @@ +"""Idle discovery must be cheap, fail visibly, and preserve explicit requests.""" +import json +from pathlib import Path +import re +from types import SimpleNamespace +import unittest +from unittest.mock import Mock + +from goal_preflight import PreflightError, has_open_goal, should_run + + +def issue(*labels, **fields): + return {"state": "open", "labels": [{"name": label} for label in labels], **fields} + + +class GoalPreflightTest(unittest.TestCase): + def test_job_filter_avoids_checkout_for_irrelevant_events(self): + source = (Path(__file__).resolve().parents[1] / "goal.md").read_text() + job = source.split(" preflight:\n", 1)[1].split("\nif:", 1)[0] + expression = re.search(r" if: >-\n(.*?) runs-on:", job, re.DOTALL).group(1) + # This condition uses only equality, boolean operators, and the two + # string/list helpers below; evaluate the actual source expression. + expression = " ".join(expression.split()).replace("||", "or").replace("&&", "and") + + def matches(event_name, **bodies): + event = SimpleNamespace(**{ + key: SimpleNamespace(body=bodies.get(key, "")) + for key in ("comment", "issue", "pull_request", "discussion") + }) + return eval(expression, {"__builtins__": {}}, { + "github": SimpleNamespace(event_name=event_name, event=event), + "contains": lambda values, value: value in values, + "fromJSON": json.loads, + "startsWith": lambda body, prefix: body.startswith(prefix), + }) + + self.assertTrue(matches("schedule")) + self.assertTrue(matches("workflow_dispatch")) + self.assertFalse(matches("push")) + self.assertFalse(matches("issue_comment", comment="ordinary comment", issue="/goal 77")) + self.assertFalse(matches("pull_request", pull_request="ordinary PR")) + for event_name, key in ( + ("issues", "issue"), ("issue_comment", "comment"), + ("pull_request", "pull_request"), ("pull_request_review_comment", "comment"), + ("discussion", "discussion"), ("discussion_comment", "comment"), + ): + for body in ("/goal", "/goal #77", "/goal\n#77"): + with self.subTest(event=event_name, body=body): + self.assertTrue(matches(event_name, **{key: body})) + + def test_manual_preflight_is_pinned_but_automatic_uses_default_branch(self): + source = (Path(__file__).resolve().parents[1] / "goal.md").read_text() + job = source.split(" preflight:\n", 1)[1].split("\nif:", 1)[0] + self.assertIn("ref: ${{ github.event_name == 'workflow_dispatch' && github.sha || github.event.repository.default_branch }}", job) + self.assertIn("persist-credentials: false", job) + self.assertIn("contents: read", job) + self.assertIn("issues: read", job) + self.assertNotIn(": write", job) + + def test_idle_schedules_and_untargeted_dispatch_skip(self): + for event_name in ("schedule", "workflow_dispatch"): + with self.subTest(event=event_name): + discover = Mock(return_value=False) + self.assertFalse(should_run(event_name, {}, discover)[0]) + discover.assert_called_once_with() + + def test_schedule_with_work_runs(self): + self.assertTrue(should_run("schedule", {}, lambda: True)[0]) + + def test_explicit_dispatch_does_not_depend_on_labels_or_api(self): + discover = Mock(side_effect=AssertionError("must not discover")) + self.assertTrue(should_run("workflow_dispatch", {"inputs": {"issue": "77"}}, discover)[0]) + discover.assert_not_called() + + def test_slash_commands_keep_the_existing_event_contract(self): + for event_name, key in ( + ("issues", "issue"), ("issue_comment", "comment"), + ("pull_request", "pull_request"), ("pull_request_review_comment", "comment"), + ("discussion", "discussion"), ("discussion_comment", "comment"), + ): + for body in ("/goal", "/goal #77", "/goal\n#77"): + with self.subTest(event=event_name, body=body): + discover = Mock(side_effect=AssertionError("must not discover")) + self.assertTrue(should_run(event_name, {key: {"body": body}}, discover)[0]) + discover.assert_not_called() + + def test_other_comments_do_not_query_or_run(self): + for body in (None, "hello", "please /goal #77", "/goalpost", "/goal\t77"): + discover = Mock(side_effect=AssertionError("must not discover")) + self.assertFalse(should_run("issue_comment", {"comment": {"body": body}}, discover)[0]) + + def test_empty_list_is_proven_idle(self): + self.assertFalse(has_open_goal("owner/repo", "test-token", get_page=lambda *args: ([], ""))) + + def test_live_goal_is_work(self): + self.assertTrue(has_open_goal("owner/repo", "test-token", + get_page=lambda *args: ([issue("goal")], ""))) + + def test_completed_closed_and_pull_requests_do_not_count(self): + entries = [issue("goal", "goal-completed"), issue("goal", state="closed"), + issue("goal", pull_request={"url": "test"}), issue("other")] + self.assertFalse(has_open_goal("owner/repo", "test-token", + get_page=lambda *args: (entries, ""))) + + def test_completed_page_does_not_hide_later_work(self): + get_page = Mock(side_effect=[ + ([issue("goal", "goal-completed")], + '; rel="next"'), + ([issue("goal")], ""), + ]) + self.assertTrue(has_open_goal("owner/repo", "test-token", get_page=get_page)) + self.assertEqual(get_page.call_count, 2) + + def test_errors_are_not_reported_as_an_empty_queue(self): + discover = Mock(side_effect=PreflightError("HTTP failure")) + with self.assertRaises(PreflightError): + should_run("schedule", {}, discover) + + def test_bad_api_shapes_fail_visibly(self): + for body in ({"message": "API rate limit exceeded"}, [None], + [{"state": "open", "labels": "goal"}], + [{"state": "open", "labels": [None]}]): + with self.subTest(body=body), self.assertRaises(PreflightError): + has_open_goal("owner/repo", "test-token", get_page=lambda *args: (body, "")) + + def test_missing_credentials_and_invalid_repository_fail(self): + for repo, token in (("owner/repo", ""), ("", "test-token"), ("../bad/repo", "test-token")): + with self.subTest(repo=repo), self.assertRaises(PreflightError): + has_open_goal(repo, token) + + def test_pagination_cannot_send_credentials_to_another_endpoint(self): + for url in ("https://attacker.example/issues", "http://api.github.com/repos/owner/repo/issues", + "https://api.github.com/repos/other/repo/issues"): + with self.subTest(url=url), self.assertRaises(PreflightError): + has_open_goal("owner/repo", "test-token", + get_page=lambda *args: ([], '<' + url + '>; rel="next"')) + + def test_repeated_page_is_an_error_not_an_infinite_loop(self): + with self.assertRaises(PreflightError): + has_open_goal("owner/repo", "test-token", get_page=lambda *args: ( + [], '; rel="next"')) + + +if __name__ == "__main__": + unittest.main() diff --git a/.github/workflows/scripts/test_goal_scheduler_errors.py b/.github/workflows/scripts/test_goal_scheduler_errors.py new file mode 100644 index 000000000..b661f0cf3 --- /dev/null +++ b/.github/workflows/scripts/test_goal_scheduler_errors.py @@ -0,0 +1,70 @@ +"""Unavailable scheduling evidence must never look like no work or no PR.""" +import json +from pathlib import Path +import tempfile +import unittest +from unittest.mock import patch +import urllib.error + +import goal_scheduler +from goal_scheduler import SchedulerError, fetch_goal_issues, find_existing_pr_for_branch + + +class GoalSchedulerErrorsTest(unittest.TestCase): + def test_http_failure_is_not_a_successful_empty_result(self): + with patch.object(goal_scheduler.urllib.request, "urlopen", + side_effect=urllib.error.URLError("unavailable")): + with self.assertRaises(SchedulerError): + goal_scheduler._http_get_json("https://api.github.com", {}) + + def test_missing_credentials_fail(self): + with self.assertRaises(SchedulerError): + fetch_goal_issues("owner/repo", "") + with self.assertRaises(SchedulerError): + find_existing_pr_for_branch("owner/repo", "goal/1-example", "") + + def test_only_a_real_empty_list_means_no_goals(self): + self.assertEqual(fetch_goal_issues("owner/repo", "test-token", + http_get_json=lambda *args: ([], "")), []) + for result in (None, {"message": "rate limited"}, [None], [{"labels": None}]): + with self.subTest(result=result), self.assertRaises(SchedulerError): + fetch_goal_issues("owner/repo", "test-token", + http_get_json=lambda *args: (result, "")) + + def test_failed_pr_lookup_cannot_create_a_duplicate_pr(self): + for result in (None, {"message": "rate limited"}, [None], [{}], [{"number": True}]): + with self.subTest(result=result), self.assertRaises(SchedulerError): + find_existing_pr_for_branch("owner/repo", "goal/1-example", "test-token", + http_get_json=lambda *args: (result, "")) + + def test_proven_no_pr_and_existing_pr_are_distinct(self): + self.assertIsNone(find_existing_pr_for_branch( + "owner/repo", "goal/1-example", "test-token", http_get_json=lambda *args: ([], ""))) + self.assertEqual(find_existing_pr_for_branch( + "owner/repo", "goal/1-example", "test-token", + http_get_json=lambda *args: ([{"number": 77}], "")), 77) + + def test_partial_issue_page_failure_is_not_partial_success(self): + pages = iter([ + ([{"labels": [{"name": "goal"}]}], + '; rel="next"'), + (None, None), + ]) + with self.assertRaises(SchedulerError): + fetch_goal_issues("owner/repo", "test-token", http_get_json=lambda *args: next(pages)) + + def test_api_failure_is_persisted_as_error_not_no_goals(self): + with tempfile.TemporaryDirectory() as directory: + output_file = Path(directory) / "goal.json" + with patch.object(goal_scheduler, "OUTPUT_DIR", directory), \ + patch.object(goal_scheduler, "OUTPUT_FILE", str(output_file)), \ + patch.object(goal_scheduler, "fetch_goal_issues", side_effect=SchedulerError("offline")): + self.assertEqual(goal_scheduler.main(), 1) + data = json.loads(output_file.read_text()) + self.assertEqual(data["error"], "offline") + self.assertFalse(data["no_goals"]) + self.assertIsNone(data["selected"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/.github/workflows/scripts/test_memory_transport.py b/.github/workflows/scripts/test_memory_transport.py new file mode 100644 index 000000000..f8b8b0e4b --- /dev/null +++ b/.github/workflows/scripts/test_memory_transport.py @@ -0,0 +1,88 @@ +"""Exercise pinned gh-aw functions, not an approximation of its glob behavior.""" +import hashlib +import json +from pathlib import Path +import re +import subprocess +import unittest + + +WORKFLOWS = Path(__file__).resolve().parents[1] +FIXTURES = Path(__file__).resolve().parent / "fixtures" / "gh-aw-v0.87.10" + + +class MemoryTransportTest(unittest.TestCase): + def test_fixture_bytes_match_the_pinned_framework(self): + for name, expected in ( + ("glob_pattern_helpers.cjs", "7f8f41db39abc0d1fdecc5a04fda79123fd7f9e1"), + ("validate_memory_files.cjs", "2fe8bf15b7611c029ce5631e27b69f9a70a96dd0"), + ): + with self.subTest(file=name): + data = (FIXTURES / name).read_bytes() + blob = b"blob " + str(len(data)).encode() + b"\0" + data + self.assertEqual(hashlib.sha1(blob).hexdigest(), expected) + + def test_actual_framework_accepts_root_markdown_without_weakening_file_types(self): + script = r''' +const assert = require("node:assert/strict"); +const fs = require("node:fs"); +const os = require("node:os"); +const path = require("node:path"); +const vm = require("node:vm"); +const fixture = process.argv[1]; +const { globPatternToRegex } = require(path.join(fixture, "glob_pattern_helpers.cjs")); +const memory = fs.mkdtempSync(path.join(os.tmpdir(), "memory-transport-")); +try { + // Exactly the invocation made by pinned push_repo_memory.cjs. + for (const pattern of ["*.md", "**/*.md"]) { + const regex = globPatternToRegex(pattern, { matchSubfolderRoot: !pattern.includes("/") }); + assert.equal(regex.test("program.md"), false); + assert.equal(regex.test("nested/program.md"), true); + } + const moduleObject = { exports: {} }; + vm.runInNewContext(fs.readFileSync(path.join(fixture, "validate_memory_files.cjs"), "utf8"), { + module: moduleObject, + require: name => name === "./error_helpers.cjs" + ? { getErrorMessage: error => error.message } + : require(name), + }); + const { validateMemoryFiles } = moduleObject.exports; + const log = { info() {}, error() {} }; + fs.mkdirSync(path.join(memory, "nested")); + fs.writeFileSync(path.join(memory, "program.md"), "# Current checkpoint\n"); + fs.writeFileSync(path.join(memory, "nested", "history.md"), "# Older checkpoint\n"); + assert.equal(validateMemoryFiles(memory, "repo", [".md"], log).valid, true); + fs.writeFileSync(path.join(memory, "unexpected.js"), "unexpected\n"); + const rejected = validateMemoryFiles(memory, "repo", [".md"], log); + assert.equal(rejected.valid, false); + assert.equal(rejected.invalidFiles.length, 1); + assert.equal(rejected.invalidFiles[0], "unexpected.js"); +} finally { + fs.rmSync(memory, { recursive: true, force: true }); +} +''' + subprocess.run(["node", "-e", script, str(FIXTURES)], check=True) + + def test_workflow_sources_use_depth_independent_markdown_allowlist(self): + for name in ("autoloop", "goal"): + with self.subTest(workflow=name): + source = (WORKFLOWS / (name + ".md")).read_text() + memory = source.split(" repo-memory:\n", 1)[1].split("\n\n", 1)[0] + self.assertIn('allowed-extensions: [".md"]', memory) + self.assertNotIn("file-glob:", memory) + + def test_generated_transport_keeps_the_same_restriction(self): + for name in ("autoloop", "goal"): + with self.subTest(workflow=name): + lock = (WORKFLOWS / (name + ".lock.yml")).read_text() + # The publication job must receive the real extension restriction, + # not a silently reintroduced compiler-default glob. + push = lock.split("id: push_repo_memory_default", 1)[1].split("\n safe_outputs:", 1)[0] + self.assertNotIn("FILE_GLOB_FILTER:", push) + allowed = re.search(r"ALLOWED_EXTENSIONS: '([^']+)'", push) + self.assertIsNotNone(allowed) + self.assertEqual(json.loads(allowed.group(1)), [".md"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/.github/workflows/wasm-verification.yml b/.github/workflows/wasm-verification.yml new file mode 100644 index 000000000..f110f2091 --- /dev/null +++ b/.github/workflows/wasm-verification.yml @@ -0,0 +1,120 @@ +name: Wasm verification + +on: + workflow_dispatch: + pull_request: + paths: + - "rust/**" + - "src/**" + - "tests/core/**" + - "tests/wasm/**" + - "tests/setup.ts" + - "package.json" + - "bun.lock" + - "bunfig.toml" + - "tsconfig.json" + - "biome.json" + - ".github/workflows/wasm-verification.yml" + push: + branches: + - main + paths: + - "rust/**" + - "src/**" + - "tests/core/**" + - "tests/wasm/**" + - "tests/setup.ts" + - "package.json" + - "bun.lock" + - "bunfig.toml" + - "tsconfig.json" + - "biome.json" + - ".github/workflows/wasm-verification.yml" + +permissions: + contents: read + +concurrency: + group: wasm-verification-${{ github.event.pull_request.number || github.ref }} + cancel-in-progress: true + +defaults: + run: + shell: bash + +jobs: + verify: + name: Build and verify real Wasm + runs-on: ubuntu-latest + timeout-minutes: 30 + steps: + - uses: actions/checkout@v4 + with: + ref: ${{ github.event.pull_request.head.sha || github.sha }} + persist-credentials: false + + - uses: oven-sh/setup-bun@v2 + with: + bun-version: "1.4.2" + + - uses: dtolnay/rust-toolchain@stable + with: + targets: wasm32-unknown-unknown + + - name: Install pinned Wasm build tooling + run: npm install --global wasm-pack@0.15.0 + + - name: Install locked dependencies + run: bun install --frozen-lockfile + + - name: Record source and toolchain + run: | + mkdir -p "$RUNNER_TEMP/wasm-verification" + { + git rev-parse HEAD + bun --version + rustc --version + cargo --version + wasm-pack --version + } | tee "$RUNNER_TEMP/wasm-verification/provenance.txt" + + - name: Rust unit tests + run: cargo test --locked --manifest-path rust/Cargo.toml 2>&1 | tee "$RUNNER_TEMP/wasm-verification/rust-tests.txt" + + - name: Verify missing Wasm cannot pass parity + run: | + if [ -d rust/pkg ]; then + mv rust/pkg "$RUNNER_TEMP/committed-wasm-pkg" + fi + if bun test ./tests/wasm/parity.test.ts > "$RUNNER_TEMP/wasm-verification/missing-module.txt" 2>&1; then + echo "::error::Wasm parity passed without its required module." + exit 1 + fi + grep -F "WASM parity requires a loaded module" "$RUNNER_TEMP/wasm-verification/missing-module.txt" + + - name: Build Wasm from source, without committed binaries + run: | + wasm-pack build --target nodejs --out-dir pkg rust/ -- --locked 2>&1 | tee "$RUNNER_TEMP/wasm-verification/build.txt" + sha256sum rust/pkg/tsb_wasm_bg.wasm | tee "$RUNNER_TEMP/wasm-verification/wasm-sha256.txt" + + - name: Required Wasm parity tests + run: bun test ./tests/wasm/ 2>&1 | tee "$RUNNER_TEMP/wasm-verification/parity-tests.txt" + + - name: Core regressions + run: bun test ./tests/core/ 2>&1 | tee "$RUNNER_TEMP/wasm-verification/core-tests.txt" + + - name: Strict typecheck + run: bun run typecheck 2>&1 | tee "$RUNNER_TEMP/wasm-verification/typecheck.txt" + + - name: Lint + run: bun run lint 2>&1 | tee "$RUNNER_TEMP/wasm-verification/lint.txt" + + - name: Upload verification evidence + if: always() + uses: actions/upload-artifact@v4 + with: + name: wasm-verification-${{ github.run_id }}-${{ github.run_attempt }} + path: | + ${{ runner.temp }}/wasm-verification/ + rust/pkg/tsb_wasm_bg.wasm + if-no-files-found: error diff --git a/tests/wasm/parity.test.ts b/tests/wasm/parity.test.ts index 47dd5d05f..07fbc8d29 100644 --- a/tests/wasm/parity.test.ts +++ b/tests/wasm/parity.test.ts @@ -22,12 +22,16 @@ let wasm: TsbWasmModule | null = null; beforeAll(async () => { wasm = await loadWasm(); + if (wasm === null) { + throw new Error("WASM parity requires a loaded module. Run `bun run wasm:build` first."); + } }); // ─── helpers ────────────────────────────────────────────────────────────────── -function skip(_label: string): void { - // no-op: caller already returns early when wasm is null +function failMissingWasm(label: string): void { + // Keep the per-test null guards type-safe, but never report missing WASM as a pass. + throw new Error(`WASM parity cannot run without its module: ${label}`); } // ─── searchsorted_f64 ──────────────────────────────────────────────────────── @@ -37,7 +41,7 @@ describe("searchsorted_f64 parity", () => { test("left insertion at boundary", () => { if (wasm === null) { - skip("left insertion at boundary"); + failMissingWasm("left insertion at boundary"); return; } const arr = new Float64Array(sortedNums); @@ -48,7 +52,7 @@ describe("searchsorted_f64 parity", () => { test("right insertion with duplicates", () => { if (wasm === null) { - skip("right insertion with duplicates"); + failMissingWasm("right insertion with duplicates"); return; } const dups = [1.0, 2.0, 3.0, 3.0, 4.0]; @@ -58,7 +62,7 @@ describe("searchsorted_f64 parity", () => { test("value less than all elements", () => { if (wasm === null) { - skip("value less than all elements"); + failMissingWasm("value less than all elements"); return; } const arr = new Float64Array(sortedNums); @@ -67,7 +71,7 @@ describe("searchsorted_f64 parity", () => { test("value greater than all elements", () => { if (wasm === null) { - skip("value greater than all elements"); + failMissingWasm("value greater than all elements"); return; } const arr = new Float64Array(sortedNums); @@ -76,7 +80,7 @@ describe("searchsorted_f64 parity", () => { test("NaN in sorted array — NaN treated as larger than all", () => { if (wasm === null) { - skip("NaN in sorted array"); + failMissingWasm("NaN in sorted array"); return; } const withNaN = [1.0, 2.0, Number.NaN]; @@ -89,7 +93,7 @@ describe("searchsorted_f64 parity", () => { test("empty array", () => { if (wasm === null) { - skip("empty array"); + failMissingWasm("empty array"); return; } const arr = new Float64Array([]); @@ -102,7 +106,7 @@ describe("searchsorted_f64 parity", () => { describe("searchsorted_many_f64 parity", () => { test("multiple values in a sorted numeric array", () => { if (wasm === null) { - skip("multiple values"); + failMissingWasm("multiple values"); return; } const sorted = [1.0, 2.0, 3.0, 4.0, 5.0]; @@ -116,7 +120,7 @@ describe("searchsorted_many_f64 parity", () => { test("right side", () => { if (wasm === null) { - skip("right side"); + failMissingWasm("right side"); return; } const sorted = [1.0, 1.0, 2.0, 3.0]; @@ -134,7 +138,7 @@ describe("searchsorted_many_f64 parity", () => { describe("argsort_f64 parity", () => { test("ascending numeric sort", () => { if (wasm === null) { - skip("ascending numeric sort"); + failMissingWasm("ascending numeric sort"); return; } const arr = [3.0, 1.0, 4.0, 1.0, 5.0, 9.0, 2.0]; @@ -145,7 +149,7 @@ describe("argsort_f64 parity", () => { test("already sorted", () => { if (wasm === null) { - skip("already sorted"); + failMissingWasm("already sorted"); return; } const arr = [1.0, 2.0, 3.0]; @@ -155,7 +159,7 @@ describe("argsort_f64 parity", () => { test("NaN placed last", () => { if (wasm === null) { - skip("NaN placed last"); + failMissingWasm("NaN placed last"); return; } const arr = [2.0, Number.NaN, 1.0]; @@ -166,7 +170,7 @@ describe("argsort_f64 parity", () => { test("single element", () => { if (wasm === null) { - skip("single element"); + failMissingWasm("single element"); return; } expect(Array.from(wasm.argsort_f64(new Float64Array([42.0])))).toEqual([0]); @@ -174,7 +178,7 @@ describe("argsort_f64 parity", () => { test("empty array", () => { if (wasm === null) { - skip("empty array"); + failMissingWasm("empty array"); return; } expect(Array.from(wasm.argsort_f64(new Float64Array([])))).toEqual([]); @@ -188,7 +192,7 @@ describe("searchsorted_str parity", () => { test("left insertion", () => { if (wasm === null) { - skip("left insertion"); + failMissingWasm("left insertion"); return; } expect(wasm.searchsorted_str([...sortedStrs], "cherry", false)).toBe( @@ -198,7 +202,7 @@ describe("searchsorted_str parity", () => { test("value not in array — between elements", () => { if (wasm === null) { - skip("value not in array"); + failMissingWasm("value not in array"); return; } expect(wasm.searchsorted_str([...sortedStrs], "avocado", false)).toBe( @@ -208,7 +212,7 @@ describe("searchsorted_str parity", () => { test("value past end", () => { if (wasm === null) { - skip("value past end"); + failMissingWasm("value past end"); return; } expect(wasm.searchsorted_str([...sortedStrs], "zucchini", false)).toBe( @@ -222,7 +226,7 @@ describe("searchsorted_str parity", () => { describe("argsort_str parity", () => { test("unsorted string array", () => { if (wasm === null) { - skip("unsorted string array"); + failMissingWasm("unsorted string array"); return; } const arr = ["cherry", "apple", "banana", "date"]; @@ -249,7 +253,7 @@ describe("nat_compare parity", () => { for (const [a, b] of cases) { test(`natCompare("${a}", "${b}")`, () => { if (wasm === null) { - skip(`natCompare("${a}", "${b}")`); + failMissingWasm(`natCompare("${a}", "${b}")`); return; } const wasmSign = Math.sign(wasm.nat_compare(a, b, false, false)); @@ -260,7 +264,7 @@ describe("nat_compare parity", () => { test("ignoreCase parity", () => { if (wasm === null) { - skip("ignoreCase parity"); + failMissingWasm("ignoreCase parity"); return; } const wasmSign = Math.sign(wasm.nat_compare("Apple", "apple", true, false)); @@ -270,7 +274,7 @@ describe("nat_compare parity", () => { test("reverse parity", () => { if (wasm === null) { - skip("reverse parity"); + failMissingWasm("reverse parity"); return; } const wasmFwd = wasm.nat_compare("file10", "file9", false, false); @@ -284,7 +288,7 @@ describe("nat_compare parity", () => { describe("nat_sorted parity", () => { test("natural sort of file names", () => { if (wasm === null) { - skip("natural sort of file names"); + failMissingWasm("natural sort of file names"); return; } const arr = ["file10", "file2", "file1", "file20"]; @@ -295,7 +299,7 @@ describe("nat_sorted parity", () => { test("reverse=true", () => { if (wasm === null) { - skip("reverse=true"); + failMissingWasm("reverse=true"); return; } const arr = ["b", "a", "c"]; @@ -306,7 +310,7 @@ describe("nat_sorted parity", () => { test("ignoreCase=true", () => { if (wasm === null) { - skip("ignoreCase=true"); + failMissingWasm("ignoreCase=true"); return; } const arr = ["Banana", "apple", "Cherry"]; @@ -317,7 +321,7 @@ describe("nat_sorted parity", () => { test("empty array", () => { if (wasm === null) { - skip("empty array"); + failMissingWasm("empty array"); return; } expect(wasm.nat_sorted([], false, false)).toEqual([]); @@ -325,7 +329,7 @@ describe("nat_sorted parity", () => { test("single element", () => { if (wasm === null) { - skip("single element"); + failMissingWasm("single element"); return; } expect(wasm.nat_sorted(["x"], false, false)).toEqual(["x"]); @@ -337,7 +341,7 @@ describe("nat_sorted parity", () => { describe("nat_argsort parity", () => { test("argsort matches natSorted order", () => { if (wasm === null) { - skip("argsort matches natSorted order"); + failMissingWasm("argsort matches natSorted order"); return; } const arr = ["file10", "file2", "file1"]; @@ -348,7 +352,7 @@ describe("nat_argsort parity", () => { test("reverse=true", () => { if (wasm === null) { - skip("reverse=true"); + failMissingWasm("reverse=true"); return; } const arr = ["a", "c", "b"]; @@ -359,7 +363,7 @@ describe("nat_argsort parity", () => { test("ignoreCase=true", () => { if (wasm === null) { - skip("ignoreCase=true"); + failMissingWasm("ignoreCase=true"); return; } const arr = ["Banana", "apple", "Cherry"]; @@ -374,7 +378,7 @@ describe("nat_argsort parity", () => { describe("sum_f64 parity", () => { test("basic sum", () => { if (wasm === null) { - skip("basic sum"); + failMissingWasm("basic sum"); return; } const arr = new Float64Array([1.0, 2.0, 3.0, 4.0, 5.0]); @@ -383,7 +387,7 @@ describe("sum_f64 parity", () => { test("NaN values are skipped", () => { if (wasm === null) { - skip("NaN skipped"); + failMissingWasm("NaN skipped"); return; } const arr = new Float64Array([1.0, Number.NaN, 3.0]); @@ -392,7 +396,7 @@ describe("sum_f64 parity", () => { test("empty array returns 0", () => { if (wasm === null) { - skip("empty"); + failMissingWasm("empty"); return; } expect(wasm.sum_f64(new Float64Array([]))).toBe(0.0); @@ -400,7 +404,7 @@ describe("sum_f64 parity", () => { test("all-NaN returns 0", () => { if (wasm === null) { - skip("all-NaN"); + failMissingWasm("all-NaN"); return; } expect(wasm.sum_f64(new Float64Array([Number.NaN, Number.NaN]))).toBe(0.0); @@ -410,7 +414,7 @@ describe("sum_f64 parity", () => { describe("mean_f64 parity", () => { test("basic mean", () => { if (wasm === null) { - skip("basic mean"); + failMissingWasm("basic mean"); return; } const arr = new Float64Array([1.0, 2.0, 3.0, 4.0, 5.0]); @@ -419,7 +423,7 @@ describe("mean_f64 parity", () => { test("NaN values are skipped", () => { if (wasm === null) { - skip("NaN skipped"); + failMissingWasm("NaN skipped"); return; } const arr = new Float64Array([1.0, Number.NaN, 5.0]); @@ -428,7 +432,7 @@ describe("mean_f64 parity", () => { test("empty array returns NaN", () => { if (wasm === null) { - skip("empty"); + failMissingWasm("empty"); return; } expect(wasm.mean_f64(new Float64Array([]))).toBeNaN(); @@ -438,7 +442,7 @@ describe("mean_f64 parity", () => { describe("min_f64 parity", () => { test("basic min", () => { if (wasm === null) { - skip("basic min"); + failMissingWasm("basic min"); return; } const arr = new Float64Array([3.0, 1.0, 4.0, 1.5]); @@ -447,7 +451,7 @@ describe("min_f64 parity", () => { test("NaN values are skipped", () => { if (wasm === null) { - skip("NaN skipped"); + failMissingWasm("NaN skipped"); return; } expect(wasm.min_f64(new Float64Array([Number.NaN, 2.0, 1.0]))).toBe(1.0); @@ -455,7 +459,7 @@ describe("min_f64 parity", () => { test("empty array returns NaN", () => { if (wasm === null) { - skip("empty"); + failMissingWasm("empty"); return; } expect(wasm.min_f64(new Float64Array([]))).toBeNaN(); @@ -465,7 +469,7 @@ describe("min_f64 parity", () => { describe("max_f64 parity", () => { test("basic max", () => { if (wasm === null) { - skip("basic max"); + failMissingWasm("basic max"); return; } const arr = new Float64Array([3.0, 1.0, 4.0, 1.5]); @@ -474,7 +478,7 @@ describe("max_f64 parity", () => { test("NaN values are skipped", () => { if (wasm === null) { - skip("NaN skipped"); + failMissingWasm("NaN skipped"); return; } expect(wasm.max_f64(new Float64Array([Number.NaN, 2.0, 5.0]))).toBe(5.0); @@ -482,7 +486,7 @@ describe("max_f64 parity", () => { test("empty array returns NaN", () => { if (wasm === null) { - skip("empty"); + failMissingWasm("empty"); return; } expect(wasm.max_f64(new Float64Array([]))).toBeNaN(); @@ -492,7 +496,7 @@ describe("max_f64 parity", () => { describe("var_f64 parity", () => { test("known variance (ddof=1)", () => { if (wasm === null) { - skip("known variance"); + failMissingWasm("known variance"); return; } // Variance of [2, 4, 4, 4, 5, 5, 7, 9] = 4.571... @@ -502,7 +506,7 @@ describe("var_f64 parity", () => { test("NaN values are skipped", () => { if (wasm === null) { - skip("NaN skipped"); + failMissingWasm("NaN skipped"); return; } const arr = new Float64Array([2.0, Number.NaN, 4.0]); @@ -511,7 +515,7 @@ describe("var_f64 parity", () => { test("single element returns NaN (ddof=1 makes n-1=0)", () => { if (wasm === null) { - skip("single element"); + failMissingWasm("single element"); return; } expect(wasm.var_f64(new Float64Array([5.0]), 1)).toBeNaN(); @@ -521,7 +525,7 @@ describe("var_f64 parity", () => { describe("std_f64 parity", () => { test("basic std (ddof=1)", () => { if (wasm === null) { - skip("basic std"); + failMissingWasm("basic std"); return; } const arr = new Float64Array([2.0, 4.0, 4.0, 4.0, 5.0, 5.0, 7.0, 9.0]); @@ -530,7 +534,7 @@ describe("std_f64 parity", () => { test("empty returns NaN", () => { if (wasm === null) { - skip("empty"); + failMissingWasm("empty"); return; } expect(wasm.std_f64(new Float64Array([]), 1)).toBeNaN(); @@ -540,7 +544,7 @@ describe("std_f64 parity", () => { describe("median_f64 parity", () => { test("odd count median", () => { if (wasm === null) { - skip("odd count"); + failMissingWasm("odd count"); return; } expect(wasm.median_f64(new Float64Array([3.0, 1.0, 2.0]))).toBe(2.0); @@ -548,7 +552,7 @@ describe("median_f64 parity", () => { test("even count median is average of two middle", () => { if (wasm === null) { - skip("even count"); + failMissingWasm("even count"); return; } expect(wasm.median_f64(new Float64Array([1.0, 2.0, 3.0, 4.0]))).toBe(2.5); @@ -556,7 +560,7 @@ describe("median_f64 parity", () => { test("NaN values are skipped", () => { if (wasm === null) { - skip("NaN skipped"); + failMissingWasm("NaN skipped"); return; } expect(wasm.median_f64(new Float64Array([1.0, Number.NaN, 3.0]))).toBe(2.0); @@ -564,7 +568,7 @@ describe("median_f64 parity", () => { test("empty returns NaN", () => { if (wasm === null) { - skip("empty"); + failMissingWasm("empty"); return; } expect(wasm.median_f64(new Float64Array([]))).toBeNaN(); @@ -576,7 +580,7 @@ describe("median_f64 parity", () => { describe("rolling_sum_f64 parity", () => { test("basic rolling sum (window=3)", () => { if (wasm === null) { - skip("rolling sum"); + failMissingWasm("rolling sum"); return; } const arr = new Float64Array([1.0, 2.0, 3.0, 4.0, 5.0]); @@ -590,7 +594,7 @@ describe("rolling_sum_f64 parity", () => { test("min_periods=1 produces earlier results", () => { if (wasm === null) { - skip("min_periods=1"); + failMissingWasm("min_periods=1"); return; } const arr = new Float64Array([1.0, 2.0, 3.0]); @@ -604,7 +608,7 @@ describe("rolling_sum_f64 parity", () => { describe("rolling_mean_f64 parity", () => { test("basic rolling mean (window=3)", () => { if (wasm === null) { - skip("rolling mean"); + failMissingWasm("rolling mean"); return; } const arr = new Float64Array([1.0, 2.0, 3.0, 4.0, 5.0]); @@ -622,7 +626,7 @@ describe("rolling_mean_f64 parity", () => { describe("expanding_mean_f64 parity", () => { test("cumulative mean grows correctly", () => { if (wasm === null) { - skip("expanding mean"); + failMissingWasm("expanding mean"); return; } const arr = new Float64Array([1.0, 3.0, 5.0]); @@ -636,7 +640,7 @@ describe("expanding_mean_f64 parity", () => { describe("expanding_sum_f64 parity", () => { test("cumulative sum", () => { if (wasm === null) { - skip("expanding sum"); + failMissingWasm("expanding sum"); return; } const arr = new Float64Array([1.0, 2.0, 3.0]);