diff --git a/.env.example b/.env.example index e3e38c57..f03fba1f 100644 --- a/.env.example +++ b/.env.example @@ -218,9 +218,10 @@ LOG_SERVICE_NAME=vortex-backend # Sentry DSN for error reporting. Leave blank to disable Sentry entirely. SENTRY_DSN= -# Bearer token for GET /metrics (issue #298). -# Empty = unauthenticated in dev/test; production always requires a value. +# ─── Metrics access control (issue #298) ───────────────────────────────────── +# Bearer token required for GET /metrics. Empty = endpoint disabled (403). # Generate with: openssl rand -hex 32 +# Required and validated (min 16 chars) in production. METRICS_TOKEN= # Optional log shipping to a central collector. Off by default so local dev and diff --git a/.env.testnet.example b/.env.testnet.example index e69de29b..1d170111 100644 --- a/.env.testnet.example +++ b/.env.testnet.example @@ -0,0 +1,122 @@ +# .env.testnet.example +# +# Environment template for LOCAL DEVELOPMENT against Stellar TESTNET. +# Copy to .env and fill in any values marked with . +# +# cp .env.testnet.example .env +# +# Testnet is safe to experiment with — tokens have no real value and contract +# deployments are free via Friendbot. Never reuse testnet keys on mainnet. +# +# Closes #136 + +# ─── Database ──────────────────────────────────────────────────────────────── +# Local Docker Compose default. Adjust if you use a remote or managed DB. +DATABASE_URL=postgresql://vortex:vortex@localhost:5432/vortex?schema=public + +# ─── Server ────────────────────────────────────────────────────────────────── +PORT=4000 +NODE_ENV=development + +# ─── Stellar / Soroban ─────────────────────────────────────────────────────── +STELLAR_NETWORK=testnet +SOROBAN_RPC_URL=https://soroban-testnet.stellar.org + +# Testnet contract IDs — leave blank until you have deployed contracts. +# The service boots without them; on-chain write paths are no-ops when empty. +SETTLEMENT_CONTRACT_ID= +SOLVER_REGISTRY_CONTRACT_ID= + +# Testnet signing key — generate a throwaway keypair, fund it with Friendbot, +# and paste the secret seed here. Never reuse this key on mainnet. +# +# # Generate a new key: +# npx @stellar/stellar-cli keys generate local-dev --network testnet +# npx @stellar/stellar-cli keys show local-dev +# +# # Or via the SDK: +# node -e "console.log(require('@stellar/stellar-sdk').Keypair.random().secret())" +# +# # Fund it (testnet only): +# curl "https://friendbot.stellar.org/?addr=" +# +# Optional in development — leave blank to skip on-chain writes. +SOROBAN_SIGNING_KEY= + +# Fee percentile used when estimating Soroban inclusion fees. +# p50 is a safe default for testnet; raise to p90+ for time-sensitive mainnet txs. +SOROBAN_FEE_PERCENTILE=p50 + +# ─── CORS ──────────────────────────────────────────────────────────────────── +# Wildcard is fine for local development — tighten this in staging/production. +CORS_ORIGIN=* + +# ─── WebSocket ─────────────────────────────────────────────────────────────── +WS_MAX_CONNECTIONS=1000 + +# ─── Pluggable signer backend (issue #400) ─────────────────────────────────── +# SIGNER_BACKEND=local is the default for development. +# In production use SIGNER_BACKEND=vault and supply VAULT_ADDR + VAULT_TOKEN. +SIGNER_BACKEND=local +VAULT_ADDR= +VAULT_TOKEN= +VAULT_TRANSIT_KEY_NAME=vortex-signer +ALLOW_LOCAL_SIGNER_IN_PROD=false +# ─── Resource-exhaustion limits (issue #476) ───────────────────────────────── +# Maximum JSON nesting depth — rejects deeply-nested body attacks (default 10). +JSON_MAX_DEPTH=10 +# Maximum chain values in a single WS subscribe message (default 20). +WS_MAX_FILTER_CHAINS=20 +# Maximum active subscriptions per WS connection (default 10). +WS_MAX_SUBSCRIPTIONS=10 +# Postgres statement_timeout for standard queries in ms (default 5000). +DB_QUERY_TIMEOUT_MS=5000 +# Postgres statement_timeout for batch queries in ms (default 10000). +DB_BATCH_QUERY_TIMEOUT_MS=10000 +# Postgres statement_timeout for stats queries in ms (default 15000). +DB_STATS_QUERY_TIMEOUT_MS=15000 + +# Emergency kill-switch (issue #477) +# Postgres-backed so a pause survives a restart and reaches every replica. +KILLSWITCH_OPERATOR_TOKEN= +KILLSWITCH_REDIS_URL= +KILLSWITCH_POLL_MS=2000 +KILLSWITCH_PERSISTENCE=prisma + +# ─── Observability (optional) ──────────────────────────────────────────────── +# Leave blank to disable Sentry error reporting. +SENTRY_DSN= + +# ─── Metrics access control (issue #298) ───────────────────────────────────── +# Bearer token required for GET /metrics. Empty = endpoint disabled (403). +# Generate with: openssl rand -hex 32 +METRICS_TOKEN= + +# debug | info | warn | error (defaults to "debug" in development) +LOG_LEVEL=debug + +# ── Shadow-mode divergence monitor (issue #401) ───────────────────────── +# Off by default in every environment. It runs read-only `simulateTransaction` +# calls against SETTLEMENT_CONTRACT_ID in parallel with the off-chain intent +# path and never signs or submits anything. +# +# SHADOW_SOURCE_ACCOUNT only has to be a valid Stellar public key: it is used to +# populate the source-account field of the simulated envelope and is never +# signed, never charged a fee and never broadcast. It must still be set, or +# every transition reports "contract_unconfigured". +SHADOW_MODE_ENABLED=false +SHADOW_SAMPLE_RATE=1 +SHADOW_QUEUE_MAX=256 +SHADOW_CONCURRENCY=4 +SHADOW_SOURCE_ACCOUNT= +# ─── Governance / Protocol Parameters ──────────────────────────────────────── +# On-chain governance parameters contract ID — leave blank to use code defaults. +PARAMS_CONTRACT_ID= + +# Poll interval in ms. 30 000 is fine for testnet. +PARAMS_POLL_INTERVAL_MS=30000 +# ─── Leader election ───────────────────────────────────────────────────────── +# Enable for multi-replica testnet deployments. +LEADER_ELECTION_ENABLED=false +LEADER_ELECTION_HEARTBEAT_MS=5000 + diff --git a/.github/ISSUE_TEMPLATE/bug_report.md b/.github/ISSUE_TEMPLATE/bug_report.md new file mode 100644 index 00000000..074f7958 --- /dev/null +++ b/.github/ISSUE_TEMPLATE/bug_report.md @@ -0,0 +1,47 @@ +--- +name: Bug report +about: Report a reproducible defect in the vortex-backend service +title: "[Bug] " +labels: ["bug", "needs-triage"] +assignees: [] +--- + +## Description + + + +## Steps to reproduce + +1. +2. +3. + +## Expected behaviour + + + +## Actual behaviour + + + +## Environment + +| Field | Value | +|-------|-------| +| Node version | | +| `npm run build` passes? | | +| `NODE_ENV` | | +| `INTENTS_PERSISTENCE` | | +| Deployment target | | + +## Relevant logs or screenshots + + + +``` +(paste here) +``` + +## Additional context + + diff --git a/.github/ISSUE_TEMPLATE/contributor_claim.md b/.github/ISSUE_TEMPLATE/contributor_claim.md new file mode 100644 index 00000000..9e3f6190 --- /dev/null +++ b/.github/ISSUE_TEMPLATE/contributor_claim.md @@ -0,0 +1,40 @@ +--- +name: Contributor claim (Drips Wave) +about: Claim a numbered issue from issues.md to work on as part of a Drips Wave +title: "[Claim] # — " +labels: ["drips-wave", "contributor-claim"] +assignees: [] +--- + +## Issue being claimed + + + +**Issue:** # + +## Contributor + + + +**GitHub:** @ + +## Approach outline + + + +## Questions or blockers + + + +## Estimated timeline + + + +--- + +_By claiming this issue you agree to follow the [CONTRIBUTING.md](../CONTRIBUTING.md) +guidelines and the [Code of Conduct](../CODE_OF_CONDUCT.md)._ diff --git a/.github/ISSUE_TEMPLATE/feature_request.md b/.github/ISSUE_TEMPLATE/feature_request.md new file mode 100644 index 00000000..cd5a66a5 --- /dev/null +++ b/.github/ISSUE_TEMPLATE/feature_request.md @@ -0,0 +1,35 @@ +--- +name: Feature request +about: Propose a new feature or improvement for vortex-backend +title: "[Feature] " +labels: ["enhancement", "needs-triage"] +assignees: [] +--- + +## Summary + + + +## Motivation + + + +## Proposed solution + + + +## Alternatives considered + + + +## Acceptance criteria + + +- [ ] +- [ ] + +## Additional context + + diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index e53cbef6..e69de29b 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -1,1236 +0,0 @@ -name: CI - -on: - push: - branches: [main] - pull_request: - branches: [main] - -# Cancel in-progress runs for the same branch/PR -concurrency: - group: ${{ github.workflow }}-${{ github.ref }} - cancel-in-progress: true - -jobs: - secrets-scan: - name: Secrets Scanning (Gitleaks) - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - with: - fetch-depth: 0 - - name: Gitleaks Scan - uses: gitleaks/gitleaks-action@ff98106e4c7b2bc287b24eaf42907196329070c7 # v2.3.9 - env: - GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} - - vulnerability-scan: - name: Dependency Vulnerability Audit - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0 - with: - node-version: 20 - cache: npm - - run: npm audit --audit-level=high - - license-scan: - name: Dependency License Compliance - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0 - with: - node-version: 20 - cache: npm - - name: Install dependencies - run: npm ci - # Permissive-only allow-list, reconciled with the project's own - # MIT/Apache-2.0 license (issue #107). Fails the build on any - # dependency (direct or transitive) carrying a copyleft or - # unrecognized/unlicensed license. - - name: Check dependency licenses - run: | - npx license-checker --production \ - --onlyAllow 'MIT;Apache-2.0;BSD-2-Clause;BSD-3-Clause;ISC;0BSD;CC0-1.0;Unlicense' \ - --excludePrivatePackages - - commitlint: - name: Commit Message Lint - runs-on: ubuntu-latest - # Only meaningful on PRs — on direct pushes to main we skip this check - # since squash merges are handled by the branch protection rule. - if: github.event_name == 'pull_request' - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - with: - # Fetch enough history to validate all commits in the PR. - fetch-depth: 0 - - - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0 - with: - node-version: 20 - cache: npm - - - name: Install dependencies - run: npm ci - - # Lint every commit in the PR from the merge base to HEAD. - - name: Lint commit messages - run: | - npx commitlint \ - --from ${{ github.event.pull_request.base.sha }} \ - --to ${{ github.event.pull_request.head.sha }} \ - --verbose - - backend: - # The en dash in the name is not cosmetic: branch protection stores the - # exact check name as a string, so replacing it with a hyphen renames a - # required check and blocks every PR until an admin edits the rule. - name: Backend (Nest) – Node ${{ matrix.node-version }} - runs-on: ubuntu-latest - strategy: - matrix: - node-version: [20, 22] - - # Spin up a PostgreSQL service container so Prisma can connect during the - # migrate step (migrate deploy requires a live DB). - services: - postgres: - image: postgres:16-alpine - env: - POSTGRES_USER: vortex - POSTGRES_PASSWORD: vortex - POSTGRES_DB: vortex - ports: - - 5432:5432 - # Wait until postgres is healthy before the steps run. - options: >- - --health-cmd="pg_isready -U vortex" - --health-interval=10s - --health-timeout=5s - --health-retries=5 - - env: - DATABASE_URL: postgresql://vortex:vortex@localhost:5432/vortex?schema=public - - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - - # ── Node / npm cache (#130) ────────────────────────────────────────── - - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0 - with: - node-version: ${{ matrix.node-version }} - cache: npm - - - name: Install dependencies - run: npm ci - - # ── Env drift check ──────────────────────────────────────────────────── - # Fast, early-failing check that env.validation.ts, .env*.example, and - # configuration.ts's process.env reads all agree on the same variable set. - - name: Check env var drift - run: npx tsx scripts/check-env-drift.ts - - # ── Prisma ────────────────────────────────────────────────────────────── - - name: Validate Prisma schema - run: npm run db:validate - - # `npm ci` deletes node_modules, so the generated client has to be cached - # *after* the install, not before. The key includes package-lock.json as - # well as the schema: the client is also regenerated when the Prisma - # version moves, and a client generated by a different version fails at - # runtime rather than at generate time. - - name: Restore cached Prisma client - uses: actions/cache@5a3ec84eff668545956fd18022155c47e93e2684 # v4.2.3 - with: - path: node_modules/.prisma - key: prisma-${{ runner.os }}-node${{ matrix.node-version }}-${{ hashFiles('prisma/schema.prisma', 'package-lock.json') }} - - - name: Generate Prisma client - run: npm run db:generate - - # Apply all pending migrations to the CI database so the schema stays in - # sync and any broken migration SQL is caught before merge. - - name: Run database migrations - run: npm run db:migrate:prod - - # ── Build & static checks ────────────────────────────────────────────── - - name: Lint - run: npm run lint - - - name: Type-check - run: npm run typecheck - - # dist/ is keyed on everything tsc reads, so a hit means the restored - # output is what this commit's build would produce anyway. There is - # deliberately no restore-keys prefix fallback: a partially-restored - # dist/ from a different commit is exactly the stale-artifact bug this - # cache must not be able to introduce. - - name: Restore cached build output - uses: actions/cache@5a3ec84eff668545956fd18022155c47e93e2684 # v4.2.3 - with: - path: dist - key: dist-${{ runner.os }}-node${{ matrix.node-version }}-${{ hashFiles('src/**/*.ts', 'package.json', 'tsconfig.json', 'nest-cli.json') }} - - - name: Build - run: npm run build - - # Unit and E2E tests deliberately do NOT run here. They are sharded into - # the `unit-tests`, `e2e-tests` and `coverage` jobs below (issue #486) so - # the suite fans out across runners instead of serialising behind lint - # and build on both Node versions. This job's name is a required status - # check, so it stays exactly as it was and keeps meaning "the backend - # compiles and passes static analysis on Node X". - - dast: - name: DAST API scan (OWASP ZAP) - runs-on: ubuntu-latest - needs: backend - timeout-minutes: 15 - env: - DATABASE_URL: postgresql://vortex:vortex@localhost:5432/vortex?schema=public - NODE_ENV: test - PORT: 4000 - CORS_ORIGIN: "*" - services: - postgres: - image: postgres:16-alpine - env: - POSTGRES_USER: vortex - POSTGRES_PASSWORD: vortex - POSTGRES_DB: vortex - ports: - - 5432:5432 - options: >- - --health-cmd="pg_isready -U vortex" - --health-interval=10s - --health-timeout=5s - --health-retries=5 - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0 - with: - node-version: 20 - cache: npm - - name: Install dependencies - run: npm ci - - name: Generate Prisma client - run: npm run db:generate - - name: Run database migrations - run: npm run db:migrate:prod - - name: Build backend - run: npm run build - - name: Start backend for ZAP scan - run: >- - npm run start > /tmp/vortex-backend.log 2>&1 & - - name: Wait for backend to become healthy - run: | - for i in $(seq 1 60); do - if curl -fsS http://localhost:${PORT}/health >/dev/null; then - echo "Backend healthy"; exit 0; - fi - sleep 2; - done - echo "Backend never became healthy" >&2 - cat /tmp/vortex-backend.log || true - exit 1 - - name: Run OWASP ZAP API scan - run: | - docker run --rm \ - --network host \ - -v "$PWD/.github/zap:/zap/wrk:ro" \ - -v "$PWD:/zap/output:rw" \ - owasp/zap2docker-stable \ - zap-api-scan.py \ - -t http://localhost:${PORT}/docs-json \ - -f openapi \ - -m 10 \ - -J /zap/output/zap-report.json \ - -r /zap/output/zap-report.html \ - -x /zap/output/zap-report.xml \ - \ - || true - - name: Fail on medium+ findings - run: | - python - <<'PY' - import json, os, sys - path = "zap-report.json" - if not os.path.exists(path): - print("No ZAP report generated; failing the job.") - sys.exit(1) - data = json.load(open(path, encoding="utf-8")) - findings = [] - for site in data.get("site", []): - for alert in site.get("alerts", []): - risk = int(alert.get("riskcode", "0")) - if risk >= 2: - findings.append({ - "name": alert.get("name"), - "risk": alert.get("riskdesc", "Unknown"), - "url": alert.get("url", "unknown"), - }) - rules = json.load(open("zap-report.json", encoding="utf-8")) if False else None - if findings: - print("Medium+ ZAP findings detected:") - for item in findings: - print(f"- {item['name']} ({item['risk']}) @ {item['url']}") - sys.exit(1) - print("No medium+ ZAP findings detected.") - PY - - name: Convert ZAP results to SARIF - if: always() - run: | - python - <<'PY' - import json - import os - src = "zap-report.json" - out = "zap-report.sarif" - if not os.path.exists(src): - print("No ZAP results to convert; writing an empty SARIF file.") - empty = { - "$schema": "https://json.schemastore.org/sarif-2.1.0.json", - "version": "2.1.0", - "runs": [{ - "tool": {"driver": {"name": "OWASP ZAP", "informationUri": "https://www.zaproxy.org/", "rules": []}}, - "results": [] - }] - } - json.dump(empty, open(out, "w", encoding="utf-8")) - raise SystemExit(0) - data = json.load(open(src, encoding="utf-8")) - results = [] - rules = [] - seen = set() - for site in data.get("site", []): - for alert in site.get("alerts", []): - rule_id = alert.get("pluginid", alert.get("alert", "unknown")) - if rule_id not in seen: - seen.add(rule_id) - rules.append({ - "id": str(rule_id), - "shortDescription": {"text": alert.get("name", "OWASP ZAP alert")}, - "fullDescription": {"text": alert.get("desc", "Alert reported by OWASP ZAP")}, - "defaultConfiguration": {"level": "warning"}, - "helpUri": "https://www.zaproxy.org/docs/alerts/" - }) - level = "warning" - risk = alert.get("riskdesc", "Informational") - if "High" in risk: - level = "error" - elif "Medium" in risk: - level = "warning" - results.append({ - "ruleId": str(rule_id), - "level": level, - "message": {"text": alert.get("name", "OWASP ZAP alert")}, - "locations": [{"physicalLocation": {"artifactLocation": {"uri": alert.get("url", "http://localhost")}}}], - "properties": {"issue_confidence": alert.get("confidence", "Medium")} - }) - sarif = { - "$schema": "https://json.schemastore.org/sarif-2.1.0.json", - "version": "2.1.0", - "runs": [{ - "tool": {"driver": {"name": "OWASP ZAP", "informationUri": "https://www.zaproxy.org/", "rules": rules}}, - "results": results - }] - } - json.dump(sarif, open(out, "w", encoding="utf-8")) - PY - - name: Upload SARIF to GitHub code scanning - if: always() - uses: github/codeql-action/upload-sarif@v3 - with: - sarif_file: zap-report.sarif - - # ── Sharded test fan-out (issue #486) ─────────────────────────────────── - # Three jobs replace the old `npm test` + `npm run test:e2e` pair: - # unit-tests -> N Jest shards, each writing its own coverage + flake report - # e2e-tests -> M Jest shards against the migrated Postgres service - # coverage -> merges the shard reports and enforces the 70% gate - # - # fail-fast is off on both matrices on purpose: with it on, the first failing - # shard cancels its siblings, their coverage artifacts are never uploaded, - # and the merge job reports "a shard never wrote its report" instead of the - # real failure. Every shard has to finish so the coverage merge is complete. - - unit-tests: - name: Unit tests (shard ${{ matrix.shard }}/${{ matrix.shard-total }}) - runs-on: ubuntu-latest - needs: backend - strategy: - fail-fast: false - matrix: - shard: [1, 2, 3, 4] - shard-total: [4] - services: - postgres: - image: postgres:16-alpine - env: - POSTGRES_USER: vortex - POSTGRES_PASSWORD: vortex - POSTGRES_DB: vortex - ports: - - 5432:5432 - options: >- - --health-cmd="pg_isready -U vortex" - --health-interval=10s - --health-timeout=5s - --health-retries=5 - env: - DATABASE_URL: postgresql://vortex:vortex@localhost:5432/vortex?schema=public - # Opts the Postgres repository contract suite in - # (src/intents/prisma-intents.repository.spec.ts — issue #404). - TEST_DATABASE_URL: postgresql://vortex:vortex@localhost:5432/vortex?schema=public - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - - - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0 - with: - # Tests run on the lowest supported Node only. Running the same 35 - # spec files on 20 and 22 doubles CI minutes to re-test runtime - # behaviour the build job already proves compiles on both; the - # version-specific risk is a native module, which `npm run build` - # loads on both anyway. - node-version: 20 - cache: npm - - - name: Install dependencies - run: npm ci - - # Same Prisma client cache as the backend job: generate-and-migrate is - # the slowest fixed cost in a test shard, and the key is content-addressed - # so a schema or lockfile change invalidates it on its own. - - name: Restore cached Prisma client - uses: actions/cache@5a3ec84eff668545956fd18022155c47e93e2684 # v4.2.3 - with: - path: node_modules/.prisma - key: prisma-${{ runner.os }}-node20-${{ hashFiles('prisma/schema.prisma', 'package-lock.json') }} - - - name: Generate Prisma client - run: npm run db:generate - - - name: Run database migrations - run: npm run db:migrate:prod - - # run-tests.mjs applies test/quarantine.json, writes the shard's - # Istanbul report to its own directory (so shards cannot clobber each - # other), and retries only the failing files to separate a flake from a - # real failure. - - name: Run unit test shard - run: >- - node scripts/ci/run-tests.mjs - --suite=unit - --shard=${{ matrix.shard }}/${{ matrix.shard-total }} - --out-dir=ci-artifacts/unit-s${{ matrix.shard }} - --coverage-dir=coverage-shards/unit-s${{ matrix.shard }} - - - name: Upload shard coverage - if: always() - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 - with: - name: coverage-unit-s${{ matrix.shard }} - path: coverage-shards/unit-s${{ matrix.shard }} - if-no-files-found: warn - - - name: Upload flake report and Jest results - if: always() - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 - with: - name: unit-s${{ matrix.shard }}-logs - path: ci-artifacts/unit-s${{ matrix.shard }} - if-no-files-found: warn - - e2e-tests: - name: E2E tests (shard ${{ matrix.shard }}/${{ matrix.shard-total }}) - runs-on: ubuntu-latest - needs: backend - strategy: - fail-fast: false - matrix: - shard: [1, 2] - shard-total: [2] - services: - postgres: - image: postgres:16-alpine - env: - POSTGRES_USER: vortex - POSTGRES_PASSWORD: vortex - POSTGRES_DB: vortex - ports: - - 5432:5432 - options: >- - --health-cmd="pg_isready -U vortex" - --health-interval=10s - --health-timeout=5s - --health-retries=5 - env: - DATABASE_URL: postgresql://vortex:vortex@localhost:5432/vortex?schema=public - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - - - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0 - with: - node-version: 20 - cache: npm - - - name: Install dependencies - run: npm ci - - # Same Prisma client cache as the backend job: generate-and-migrate is - # the slowest fixed cost in a test shard, and the key is content-addressed - # so a schema or lockfile change invalidates it on its own. - - name: Restore cached Prisma client - uses: actions/cache@5a3ec84eff668545956fd18022155c47e93e2684 # v4.2.3 - with: - path: node_modules/.prisma - key: prisma-${{ runner.os }}-node20-${{ hashFiles('prisma/schema.prisma', 'package-lock.json') }} - - - name: Generate Prisma client - run: npm run db:generate - - - name: Run database migrations - run: npm run db:migrate:prod - - # Anvil for the EVM deposit-verification integration test (issue #403, - # test/evm/deposit-verifier.anvil.test.ts — skipped when anvil is absent). - - name: Install Foundry (anvil) - uses: foundry-rs/foundry-toolchain@v1 - - # No --coverage-dir here. test/jest-e2e.json has no collectCoverageFrom, - # so its coverage map has different file keys than the unit config's; - # merging the two would trip the shard-consistency check in - # coverage-merge.mjs. The coverage gate is measured on the unit shards, - # which is also where nearly all of the code is exercised. - - name: Run E2E test shard - run: >- - node scripts/ci/run-tests.mjs - --suite=e2e - --shard=${{ matrix.shard }}/${{ matrix.shard-total }} - --out-dir=ci-artifacts/e2e-s${{ matrix.shard }} - - - name: Upload flake report and Jest results - if: always() - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 - with: - name: e2e-s${{ matrix.shard }}-logs - path: ci-artifacts/e2e-s${{ matrix.shard }} - if-no-files-found: warn - - # ── E2E against the Postgres-backed intents store (issue #404) ───────────── - # The e2e-tests shards use the default in-memory store. This runs the same - # suite, including the concurrency and POST /intents latency load tests, - # with INTENTS_STORE=postgres. --runInBand: suites share one database, so - # parallel workers would race on table-wide counts (e.g. the stats deltas). - e2e-postgres-store: - name: E2E tests (INTENTS_STORE=postgres) - runs-on: ubuntu-latest - needs: backend - services: - postgres: - image: postgres:16-alpine - env: - POSTGRES_USER: vortex - POSTGRES_PASSWORD: vortex - POSTGRES_DB: vortex - ports: - - 5432:5432 - options: >- - --health-cmd="pg_isready -U vortex" - --health-interval=10s - --health-timeout=5s - --health-retries=5 - env: - DATABASE_URL: postgresql://vortex:vortex@localhost:5432/vortex?schema=public - INTENTS_STORE: postgres - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - - - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0 - with: - node-version: 20 - cache: npm - - - name: Install dependencies - run: npm ci - - - name: Generate Prisma client - run: npm run db:generate - - - name: Run database migrations - run: npm run db:migrate:prod - - - name: Install Foundry (anvil) - uses: foundry-rs/foundry-toolchain@v1 - - - name: E2E tests - run: npm run test:e2e -- --runInBand - - coverage: - name: Coverage merge and gate - runs-on: ubuntu-latest - needs: [unit-tests, e2e-tests] - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - - - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0 - with: - node-version: 20 - cache: npm - - - name: Install dependencies - run: npm ci - - # Without merge-multiple, each artifact lands in its own subdirectory of - # the target path, so the four shards arrive as - # coverage-shards/coverage-unit-s1..4/coverage-final.json -- exactly the - # one-directory-deep layout coverage-merge.mjs scans. - - name: Download shard coverage - uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4.3.0 - with: - pattern: coverage-unit-s* - path: coverage-shards - - - name: Download flake reports - uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4.3.0 - with: - pattern: "*-logs" - path: ci-artifacts - - # The 70% gate is enforced here, on the union, and never per shard: a - # shard that runs a ninth of the suite cannot meet a global threshold. - # The threshold itself is read from jest.config.js, so this stays a single - # definition rather than a second copy in the workflow. - - name: Merge shard coverage and enforce threshold - run: >- - node scripts/ci/coverage-merge.mjs - --shards=coverage-shards - --out=coverage - --expect-shards=4 - --title="Coverage (merged from 4 unit test shards)" - - - name: Upload merged coverage - if: always() - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 - with: - name: coverage-merged - path: coverage - if-no-files-found: warn - - # Pure Node, no npm ci: the quarantine file is the one piece of test - # infrastructure that can be validated before dependencies are installed, so - # it is also usable as a pre-commit guard (scripts/ci/check-quarantine.mjs). - test-hygiene: - name: Quarantine file hygiene - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - - - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0 - with: - node-version: 20 - - # Fails on a missing owner/issue, a path that no longer exists, a - # duplicate entry, or an entry older than staleAfterDays. - - name: Validate test/quarantine.json - run: node scripts/ci/check-quarantine.mjs - - # ── TLA+ model checking (issue #471) ───────────────────────────────────── - # Bounded TLC run on changes to the spec or the canonical state machine. - formal: - name: TLA+ model check (TLC, bounded) - runs-on: ubuntu-latest - if: >- - github.event_name == 'pull_request' && - (contains(join(github.event.pull_request.labels.*.name, ','), 'formal') || - true) - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - - name: Run TLC (bounded model) - uses: docker://tlaplus/tlaplus:latest - with: - args: tlc -config formal/IntentLifecycle.cfg formal/IntentLifecycle.tla - - name: Upload TLC output - if: always() - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 - with: - name: tlc-output - path: formal/*.out - - # ── Semgrep custom rules (issue #478) ──────────────────────────────────── - semgrep: - name: Semgrep custom rules - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - - uses: returntocorp/semgrep-action@713efdd345f3035192eaa63f56867b88e63e4e5d # v1 - with: - config: .semgrep/rules - error: true - - name: Semgrep rule tests - run: | - pip install semgrep - semgrep --test --config .semgrep/rules .semgrep/tests - - # ── Prometheus rules (issue #480) ──────────────────────────────────────── - promtool: - name: promtool check + test - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - - name: Install promtool - run: | - curl -sL https://github.com/prometheus/prometheus/releases/download/v2.53.0/prometheus-2.53.0.linux-amd64.tar.gz | tar xz - sudo mv prometheus-2.53.0.linux-amd64/promtool /usr/local/bin/ - - name: Check rules - run: promtool check rules ops/prometheus/rules/vortex-slo.yml ops/prometheus/rules/vortex-canary.yml - - name: Unit-test rules - run: promtool test rules ops/prometheus/rules/vortex-slo_test.yml ops/prometheus/rules/vortex-canary_test.yml - - name: Check rule file is also a valid Prometheus config fragment - run: promtool check config ops/prometheus/prometheus.yml - - # ── Grafana dashboards (issue #481) ─────────────────────────────────────── - # Two things are checked, neither of which needs a Grafana instance: - # 1. build.mjs --check proves the committed dashboard JSON is exactly what - # the generator produces, so a hand-edited panel cannot drift. - # 2. validate.mjs proves the panels are usable: every recording rule and - # metric a panel names exists, every panel has a legend, description, - # unit and threshold set, no panel reintroduces a multi-day - # histogram_quantile, and every alert links to a dashboard that exists. - # - # Both are plain Node with no dependencies, which is the whole point of - # generating the JSON ourselves instead of adding grafonnet to the toolchain. - grafana: - name: Grafana dashboards - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0 - with: - node-version: 20 - - - name: Generated dashboards are in sync with build.mjs - run: node ops/grafana/build.mjs --check - - - name: Validate dashboards, panels and alert links - run: node ops/grafana/validate.mjs - - - name: Validate the observability Compose profile - run: docker compose --profile observability config --quiet - - - name: Upload dashboards for review - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 - with: - name: grafana-dashboards - path: ops/grafana/dashboards - - # ── Synthetic canary against the docker-compose stack (issue #496) ──────── - canary: - name: Canary – local lifecycle run (docker-compose) - runs-on: ubuntu-latest - needs: backend - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0 - with: - node-version: 20 - cache: npm - - run: npm ci - - name: Generate throwaway canary keys - run: | - node -e ' - const { Keypair } = require("@stellar/stellar-sdk"); - const u = Keypair.random(), s = Keypair.random(); - console.log(`CANARY_USER_SECRET=${u.secret()}`); - console.log(`CANARY_SOLVER_SECRET=${s.secret()}`); - console.log(`CANARY_ADDRESSES=${u.publicKey()},${s.publicKey()}`); - ' >> "$GITHUB_ENV" - - name: Start stack - run: docker compose up -d --build --wait postgres redis && docker compose up -d app - - name: Wait for API - run: timeout 300 sh -c 'until curl -sf http://localhost:4000/health/live; do sleep 5; done' - - name: Run canary once - env: - CANARY_SETTLE_ONCHAIN: "false" - run: npm run canary -- --once - - name: Stack logs - if: failure() - run: docker compose logs app - - # ── Migration rollback verification ──────────────────────────────────────── - # For every prisma/migrations/*/down.sql, applies migration.sql then - # down.sql against a fresh Postgres and asserts the schema returns to its - # pre-migration (empty) state. See prisma/migrations/README.md. - migration-rollback: - name: Migration rollback verification - runs-on: ubuntu-latest - services: - postgres: - image: postgres:16-alpine - env: - POSTGRES_USER: vortex - POSTGRES_PASSWORD: vortex - POSTGRES_DB: vortex_rollback_test - ports: - - 5432:5432 - options: >- - --health-cmd pg_isready - --health-interval 10s - --health-timeout 5s - --health-retries 5 - env: - PGHOST: localhost - PGPORT: 5432 - PGUSER: vortex - PGPASSWORD: vortex - PGDATABASE: vortex_rollback_test - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - - - name: Apply each migration's up then down script and verify schema is restored - run: | - set -euo pipefail - for dir in prisma/migrations/*/; do - name=$(basename "$dir") - up="$dir/migration.sql" - down="$dir/down.sql" - [ -f "$down" ] || continue - - echo "::group::$name" - before=$(psql -Atc "select tablename from pg_tables where schemaname='public' order by 1") - psql -v ON_ERROR_STOP=1 -f "$up" - psql -v ON_ERROR_STOP=1 -f "$down" - after=$(psql -Atc "select tablename from pg_tables where schemaname='public' order by 1") - - if [ "$before" != "$after" ]; then - echo "Schema after down.sql does not match pre-migration state for $name" - echo "before: $before" - echo "after: $after" - exit 1 - fi - echo "::endgroup::" - done - - # ── Migration lint ───────────────────────────────────────────────────────── - # Blocks unsafe DDL in migrations this change adds or modifies: non-concurrent - # index builds/drops, column type rewrites, NOT NULL without a default, - # lock-heavy constraints, LOCK TABLE. Also requires a down.sql in every such - # migration. Untouched migrations aren't linted, so older ones can't fail - # retroactively. The rules and the `-- squawk-ignore` override are in - # prisma/migrations/README.md. - migration-lint: - name: Migration lint (squawk) - runs-on: ubuntu-latest - timeout-minutes: 5 - steps: - - uses: actions/checkout@v4 - with: - # Full history so the diff base below is always present. - fetch-depth: 0 - - - uses: actions/setup-node@v4 - with: - node-version: 20 - cache: npm - - # --ignore-scripts skips Prisma's engine download and the husky hook; - # neither is needed here, and it keeps the job well under 2 minutes. - - name: Install dependencies - run: npm ci --ignore-scripts - - - name: Test the migration checker (fixtures) - run: npm run test:scripts - - - name: Lint changed migrations - env: - EVENT_NAME: ${{ github.event_name }} - BEFORE_SHA: ${{ github.event.before }} - run: | - # Pull requests: GitHub checks out a merge commit whose first parent is - # the base branch tip, so HEAD^1 is exactly "everything this PR adds". - # Pushes: diff against the commit that was on the branch before the push. - if [ "$EVENT_NAME" = "pull_request" ] || [ -z "$BEFORE_SHA" ] || [ -z "${BEFORE_SHA//0/}" ]; then - base="HEAD^1" - else - base="$BEFORE_SHA" - fi - npm run check:migrations -- --base "$base" - - # ── N-1 app against N schema (issue #497) ───────────────────────────────── - # The issue's compatibility criterion: "previous app version's e2e smoke - # suite runs against the migrated schema in CI". Concretely, this job: - # - # 1. applies the migrations of THIS revision (schema N) to a fresh Postgres; - # 2. checks out the previous revision of the app (N-1) into a second - # directory and installs its own dependencies there; - # 3. runs three things from N-1 against schema N: - # a. `prisma migrate deploy` — N-1's migration set must reconcile - # against the N database. This is the exact "misordered migration" - # failure the issue describes: if a migration was rewritten, - # renamed or re-ordered between N-1 and N, this fails here instead - # of in production; - # b. a raw query through N-1's *generated Prisma client* — the e2e - # harness (test/utils/create-test-app.ts) deliberately MOCKS - # PrismaService, so the smoke specs below boot the N-1 app but do - # not touch the database. This probe is what makes the run - # schema-sensitive; - # c. N-1's e2e smoke subset (health + OpenAPI contract), when those - # specs exist at that revision. - # - # Which revision counts as "previous"? Chosen per event so that N-1 is the - # app that is actually running/deployed, not an arbitrary ancestor: - # * push: `github.event.before` — literally the tip that was on - # the branch before this push, i.e. what staging runs. On - # main that is the deployed revision by definition. - # (`HEAD^1` would be wrong for multi-commit pushes: it - # points *inside* the pushed range, at something never - # deployed.) - # * pull_request: the merge commit's first parent, i.e. the base-branch - # tip this PR is merged onto — what the branch runs. Only - # if that is not an ancestor of origin/main (checkout fell - # back to the head commit because the merge ref was - # unavailable) do we fall back to the merge-base with - # origin/main; an older base only makes the test stricter, - # never weaker. - # * first push / unreachable object: skip gracefully with an explicit - # ::notice:: — there is no N-1 to test, and failing the build for that - # would block the very first commit of a branch. - migration-compat: - name: Migration compatibility (N-1 app vs N schema) - runs-on: ubuntu-latest - # Hard budget: the issue asks for "under ~10 minutes". The two installs - # (current + N-1 worktree) dominate; everything else is seconds. - timeout-minutes: 10 - services: - postgres: - image: postgres:16-alpine - env: - POSTGRES_USER: vortex - POSTGRES_PASSWORD: vortex - POSTGRES_DB: vortex - ports: - - 5432:5432 - options: >- - --health-cmd="pg_isready -U vortex" - --health-interval=10s - --health-timeout=5s - --health-retries=5 - env: - DATABASE_URL: postgresql://vortex:vortex@localhost:5432/vortex?schema=public - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - with: - # Full history: resolving `before`, the merge-base and the worktree - # checkout all need commits that a shallow clone would omit. - fetch-depth: 0 - - - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0 - with: - node-version: 20 - cache: npm - - - name: Install dependencies (current revision) - run: npm ci - - - name: Generate Prisma client - run: npm run db:generate - - # ── Step 1: bring the database to schema N ──────────────────────────── - - name: Apply this revision's migrations (schema N) - run: npm run db:migrate:prod - - # ── Step 2: pick the previous revision, or skip with a notice ───────── - - name: Resolve the previous revision (N-1) - id: prev - env: - EVENT_NAME: ${{ github.event_name }} - BEFORE_SHA: ${{ github.event.before }} - BASE_REF: ${{ github.base_ref }} - run: | - set -euo pipefail - prev="" - case "$EVENT_NAME" in - push) - # `before` is all-zeros on the first push of a ref and can be - # missing after a force-push; both are handled by the guards. - if [ -n "${BEFORE_SHA:-}" ] && [ "${BEFORE_SHA//0/}" != "" ] \ - && git cat-file -e "${BEFORE_SHA}^{commit}" 2>/dev/null; then - prev="$BEFORE_SHA" - fi - ;; - pull_request) - # Merge ref: HEAD has two parents, first one is the base tip. - if git rev-parse --verify --quiet HEAD^2 >/dev/null \ - && git cat-file -e "HEAD^1^{commit}" 2>/dev/null \ - && git merge-base --is-ancestor "HEAD^1" "origin/${BASE_REF}"; then - prev=$(git rev-parse "HEAD^1") - else - # Fallback: the shared ancestor with the integration branch. - prev=$(git merge-base HEAD "origin/${BASE_REF}" 2>/dev/null || true) - fi - ;; - esac - if [ -z "$prev" ] || ! git cat-file -e "${prev}^{commit}" 2>/dev/null; then - echo "available=false" >> "$GITHUB_OUTPUT" - echo "::notice::No usable previous revision for this event — skipping the N-1 compatibility check (first push of the ref, force-push, or unrelated history). Nothing to compare against, so there is nothing to fail." - exit 0 - fi - if [ "$prev" = "$(git rev-parse HEAD)" ]; then - echo "available=false" >> "$GITHUB_OUTPUT" - echo "::notice::Previous revision equals HEAD — skipping the N-1 compatibility check (degenerate history)." - exit 0 - fi - echo "sha=$prev" >> "$GITHUB_OUTPUT" - echo "available=true" >> "$GITHUB_OUTPUT" - echo "N-1 revision: $prev ($(git log -1 --format=%s "$prev" | cut -c1-72))" - - # ── Step 3: materialise N-1 in a second directory ───────────────────── - # A `git worktree` shares the object database with the checkout above - # (no second clone), while giving N-1 its own package.json, lockfile and - # node_modules — the current tree stays untouched for the schema-N side. - # It lives under $RUNNER_TEMP so it cannot be mistaken for workspace - # content by later steps or by the artifact uploaders. - - name: Check out N-1 into a separate worktree - if: steps.prev.outputs.available == 'true' - id: worktree - run: | - set -euo pipefail - N1_DIR="$RUNNER_TEMP/n1-app" - git worktree add --detach "$N1_DIR" "${{ steps.prev.outputs.sha }}" - echo "N1_DIR=$N1_DIR" >> "$GITHUB_ENV" - echo "Checked out N-1 at $N1_DIR" - - - name: Install N-1 dependencies - if: steps.prev.outputs.available == 'true' - working-directory: ${{ env.N1_DIR }} - run: npm ci --no-audit --no-fund - - # @prisma/client's postinstall normally generates the client, but that is - # an implicit behaviour of whatever Prisma version N-1 pinned; make it - # explicit (and skip it cleanly on revisions whose package.json has no - # such script yet). - - name: Generate N-1 Prisma client - if: steps.prev.outputs.available == 'true' - working-directory: ${{ env.N1_DIR }} - run: | - if node -e 'process.exit(require("./package.json").scripts?.["db:generate"] ? 0 : 1)'; then - npm run db:generate - else - echo "No db:generate script at this revision; relying on @prisma/client postinstall." - fi - - # ── Step 4: the actual compatibility probes ─────────────────────────── - # (a) N-1's migration set must reconcile with schema N — the ordering / - # rewriting guard. - - name: N-1 `prisma migrate deploy` against schema N - if: steps.prev.outputs.available == 'true' - working-directory: ${{ env.N1_DIR }} - run: npx prisma migrate deploy - - # (b) The schema-sensitive probe: query through N-1's generated client. - # Written as a file (not `node -e`) so quoting cannot eat the SQL, and - # placed inside the worktree so `require("@prisma/client")` resolves - # N-1's client, not the current revision's. - - name: Query schema N through the N-1 Prisma client - if: steps.prev.outputs.available == 'true' - working-directory: ${{ env.N1_DIR }} - run: | - set -euo pipefail - cat > compat-smoke.js <<'JS' - const { PrismaClient } = require("@prisma/client"); - (async () => { - const prisma = new PrismaClient(); - const tables = await prisma.$queryRawUnsafe( - "SELECT count(*)::int AS n FROM information_schema.tables WHERE table_schema = 'public'", - ); - const n = Number(tables[0].n); - if (n < 1) throw new Error("schema N shows no public tables"); - // The table every released revision since init depends on. - await prisma.$queryRawUnsafe('SELECT count(*)::int FROM "intents"'); - console.log(`N-1 client queried schema N: ${n} public tables, intents readable`); - await prisma.$disconnect(); - })().catch((err) => { - console.error("N-1 client cannot use schema N:", err); - process.exit(1); - }); - JS - node compat-smoke.js - - # (c) N-1's e2e smoke subset. Existence is checked per file because the - # two specs may not have existed at every historical revision — missing - # coverage at N-1 is a reason to notice, not to fail the build. - - name: Run N-1 e2e smoke suite against schema N - if: steps.prev.outputs.available == 'true' - working-directory: ${{ env.N1_DIR }} - run: | - set -euo pipefail - smoke=() - for f in test/health.e2e-spec.ts test/openapi-contract.e2e-spec.ts; do - if [ -f "$f" ]; then - smoke+=("$f") - else - echo "::notice::$f does not exist at the N-1 revision — skipping that spec." - fi - done - if [ "${#smoke[@]}" -eq 0 ]; then - echo "::notice::No e2e smoke specs exist at the N-1 revision — nothing to run." - exit 0 - fi - npx jest --config test/jest-e2e.json --runTestsByPath "${smoke[@]}" --ci - - - name: Provenance for the executed comparison - if: always() && steps.prev.outputs.available == 'true' - run: | - echo "::notice::N-1 compatibility checked: N=${GITHUB_SHA} schema vs N-1=${{ steps.prev.outputs.sha }} app." - - # ── Docker build verification (#131) ────────────────────────────────────── - # Ensures every push that changes application code doesn't silently break - # the Dockerfile. Image publishing to a registry can be wired up separately - # once registry credentials are in place. - docker: - name: Docker – build verification - runs-on: ubuntu-latest - needs: backend - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - - - name: Build Docker image - uses: docker/build-push-action@263435318d21b8e681c14492fe198d362a7d2c83 # v6.18.0 - with: - context: . - push: false - tags: vortex-backend:ci - - # ── OpenAPI drift check (issue #318) ────────────────────────────────────── - # Verifies that src/generated/openapi.json (and the api-types.ts it drives) - # still matches what the current controllers would produce. A PR that - # changes a controller DTO or route without re-running `npm run generate:client` - # will fail here before stale types ship to SDK consumers. - # - # The job runs only on pull_request events so we don't block main pushes that - # are the *result* of running the generator. The sdk-publish job below - # re-generates on tags anyway, so production is always up-to-date. - openapi-drift: - name: OpenAPI drift check - runs-on: ubuntu-latest - needs: backend - if: github.event_name == 'pull_request' - services: - postgres: - image: postgres:16-alpine - env: - POSTGRES_USER: vortex - POSTGRES_PASSWORD: vortex - POSTGRES_DB: vortex - ports: - - 5432:5432 - options: >- - --health-cmd="pg_isready -U vortex" - --health-interval=10s - --health-timeout=5s - --health-retries=5 - env: - DATABASE_URL: postgresql://vortex:vortex@localhost:5432/vortex?schema=public - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - - # ── Solver SDK (issue #446): build + publish dry-run on every PR ─────────── - solver-sdk: - name: Solver SDK – build + publish dry-run - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0 - with: - node-version: 20 - cache: npm - - - name: Install dependencies - run: npm ci - - - name: Restore cached Prisma client - uses: actions/cache@5a3ec84eff668545956fd18022155c47e93e2684 # v4.2.3 - with: - path: node_modules/.prisma - key: prisma-${{ runner.os }}-node20-${{ hashFiles('prisma/schema.prisma', 'package-lock.json') }} - - - name: Generate Prisma client - run: npm run db:generate - - - name: Run database migrations - run: npm run db:migrate:prod - - # Build first — generate:client requires dist/ to boot the throw-away app. - - name: Build - run: npm run build - - # Re-generate the client from the current controllers into a temp directory, - # then diff against the checked-in src/generated/. Any difference means - # a controller change was not followed by `npm run generate:client`. - - name: Regenerate OpenAPI client - run: npm run generate:client - - - name: Check for OpenAPI drift - run: | - if ! git diff --exit-code src/generated/; then - echo "" - echo "::error::src/generated/ is out of date. Run \`npm run generate:client\` locally," - echo "::error::commit the updated files, and push again." - echo "" - git diff src/generated/ - exit 1 - fi - echo "✅ src/generated/ is in sync with the current controllers." - - run: npm ci - - run: npm run check:version -w @vortex/solver-sdk - - run: npm run build -w @vortex/solver-sdk - - run: npm publish -w @vortex/solver-sdk --dry-run - - sdk-publish: - name: Publish generated SDK - runs-on: ubuntu-latest - needs: [backend, docker] - if: startsWith(github.ref, 'refs/tags/v') - permissions: - contents: read - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - - - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0 - with: - node-version: 20 - registry-url: https://registry.npmjs.org - - - name: Install dependencies - run: npm ci - - - name: Regenerate client SDK - run: npm run generate:client - - - name: Verify packaged SDK installs cleanly - run: | - tmpdir=$(mktemp -d) - cd "$tmpdir" - npm init -y - npm install --silent "$GITHUB_WORKSPACE/src/generated" - npm install --silent typescript @types/node - cat <<'TS' > index.ts - import type { paths, components } from '@vortex-protocol/backend-sdk'; - - type IntentList = paths['/api/v1/intents']['get']; - type Intent = components['schemas']['Intent']; - - const fallback: Intent | undefined = undefined; - void fallback; - void IntentList; - TS - npx tsc --noEmit --moduleResolution node --module commonjs --target es2020 index.ts - - - name: Publish SDK to npm - env: - NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }} - run: | - version="${GITHUB_REF_NAME#v}" - npm --prefix src/generated version "$version" --no-git-tag-version - npm --prefix src/generated publish --access public diff --git a/src/common/http-exception.filter.spec.ts b/src/common/http-exception.filter.spec.ts index 4e592268..e69de29b 100644 --- a/src/common/http-exception.filter.spec.ts +++ b/src/common/http-exception.filter.spec.ts @@ -1,164 +0,0 @@ -import { HttpException, HttpStatus, ArgumentsHost } from "@nestjs/common"; -import { HttpExceptionFilter } from "./http-exception.filter"; -import * as sentryModule from "./sentry"; -import { logger } from "./logger"; - -// ── helpers ─────────────────────────────────────────────────────────────────── - -function makeHost(json: jest.Mock, requestId?: string, setHeader?: jest.Mock): ArgumentsHost { - return { - switchToHttp: () => ({ - getResponse: () => ({ - status: (_code: number) => ({ json }), - setHeader: setHeader ?? jest.fn(), - }), - getRequest: () => (requestId ? { requestId } : {}), - }), - } as unknown as ArgumentsHost; -} - -// ── Sentry mock ─────────────────────────────────────────────────────────────── -// Spy on captureException so we can assert it is called only in the 500 branch. - -jest.mock("./sentry", () => ({ - initSentry: jest.fn(), - captureException: jest.fn(), -})); - -// ── tests ───────────────────────────────────────────────────────────────────── - -describe("HttpExceptionFilter", () => { - let filter: HttpExceptionFilter; - let json: jest.Mock; - - beforeEach(() => { - filter = new HttpExceptionFilter(); - json = jest.fn(); - jest.clearAllMocks(); - }); - - describe("HttpException branch", () => { - it("returns the correct status for a 404", () => { - const host = makeHost(json); - filter.catch(new HttpException("Not found", HttpStatus.NOT_FOUND), host); - expect(json).toHaveBeenCalledWith({ error: "Not found" }); - }); - - it("does NOT call captureException for expected HttpExceptions (no alert fatigue)", () => { - const host = makeHost(json); - filter.catch(new HttpException("Bad request", HttpStatus.BAD_REQUEST), host); - expect(sentryModule.captureException).not.toHaveBeenCalled(); - }); - - it("surfaces class-validator array messages", () => { - const host = makeHost(json); - const body = { message: ["field must not be empty"], error: "Bad Request", statusCode: 400 }; - filter.catch(new HttpException(body, HttpStatus.BAD_REQUEST), host); - expect(json).toHaveBeenCalledWith({ - error: "Validation failed", - details: ["field must not be empty"], - }); - }); - }); - - describe("generic Error branch (500)", () => { - it("returns 500 json for an unhandled Error", () => { - const host = makeHost(json); - filter.catch(new Error("boom"), host); - expect(json).toHaveBeenCalledWith({ error: "boom" }); - }); - - it("calls captureException with the Error instance (Sentry alerting)", () => { - const host = makeHost(json); - const err = new Error("unexpected failure"); - filter.catch(err, host); - expect(sentryModule.captureException).toHaveBeenCalledTimes(1); - expect(sentryModule.captureException).toHaveBeenCalledWith(err); - }); - - it("wraps non-Error throws and still calls captureException", () => { - const host = makeHost(json); - filter.catch("string throw", host); - expect(sentryModule.captureException).toHaveBeenCalledTimes(1); - const captured = (sentryModule.captureException as jest.Mock).mock.calls[0][0]; - expect(captured).toBeInstanceOf(Error); - }); - - it("returns 500 with fallback message for unknown throws", () => { - const host = makeHost(json); - filter.catch(null, host); - expect(json).toHaveBeenCalledWith({ error: "Unknown error" }); - }); - - it("propagates an express error's own numeric status instead of masking with 500", () => { - const host = makeHost(json); - const payloadTooLarge = Object.assign(new Error("request entity too large"), { status: 413 }); - - filter.catch(payloadTooLarge, host); - - expect(json).toHaveBeenCalledWith({ error: "request entity too large" }); - }); - }); - - describe("requestId propagation", () => { - it("echoes requestId back on an HttpException when the request has one", () => { - const host = makeHost(json, "req-abc-123"); - - filter.catch(new HttpException("Not found", HttpStatus.NOT_FOUND), host); - - expect(json).toHaveBeenCalledWith({ error: "Not found", requestId: "req-abc-123" }); - }); - - it("omits requestId entirely when the request has none", () => { - const host = makeHost(json); - - filter.catch(new HttpException("Not found", HttpStatus.NOT_FOUND), host); - - expect(json).toHaveBeenCalledWith({ error: "Not found" }); - expect(json.mock.calls[0][0]).not.toHaveProperty("requestId"); - }); - - it("does not add requestId to the 500 branch", () => { - const host = makeHost(json, "req-abc-123"); - - filter.catch(new Error("boom"), host); - - expect(json).toHaveBeenCalledWith({ error: "boom" }); - }); - }); - - describe("custom-shaped body passthrough (issue #304)", () => { - it("forwards allowlisted fields from a custom-shaped exception body", () => { - const host = makeHost(json); - const body = { error: "fill-check failed", intentId: "abc-123", fillAmount: "100", minDstAmount: "90" }; - filter.catch(new HttpException(body, HttpStatus.BAD_REQUEST), host); - expect(json).toHaveBeenCalledWith({ - error: "fill-check failed", - intentId: "abc-123", - fillAmount: "100", - minDstAmount: "90", - }); - }); - - it("strips unknown fields from a custom-shaped body and logs a warning", () => { - const host = makeHost(json); - // eslint-disable-next-line @typescript-eslint/no-explicit-any - const warnSpy = jest.spyOn(logger, "warn").mockImplementation((() => logger) as any); - const body = { error: "something failed", intentId: "abc-123", internalDebugField: "secret" }; - filter.catch(new HttpException(body, HttpStatus.BAD_REQUEST), host); - const response = json.mock.calls[0][0] as Record; - expect(response).not.toHaveProperty("internalDebugField"); - expect(response).toHaveProperty("error"); - expect(response).toHaveProperty("intentId"); - expect(warnSpy).toHaveBeenCalledWith(expect.stringContaining("internalDebugField")); - warnSpy.mockRestore(); - }); - - it("injects requestId into a custom-shaped body response", () => { - const host = makeHost(json, "req-xyz-456"); - const body = { error: "fill-check failed", intentId: "abc-123" }; - filter.catch(new HttpException(body, HttpStatus.BAD_REQUEST), host); - expect(json).toHaveBeenCalledWith(expect.objectContaining({ requestId: "req-xyz-456" })); - }); - }); -}); diff --git a/src/common/http-exception.filter.ts b/src/common/http-exception.filter.ts index 0cb6bdb0..e69de29b 100644 --- a/src/common/http-exception.filter.ts +++ b/src/common/http-exception.filter.ts @@ -1,130 +0,0 @@ -import { ArgumentsHost, Catch, ExceptionFilter, HttpException } from "@nestjs/common"; -import { captureException } from "./sentry"; -import { logger } from "./logger"; - -/** - * Fields that are explicitly allowed to pass through in a custom-shaped exception - * body (i.e. the `b.error && !b.statusCode` branch). - * - * Any key NOT in this set will be stripped and a warning emitted in development - * so the author learns about the leak before it reaches production. In - * production the field is silently dropped so no internal detail escapes. - * - * Today's only known usage is IntentsController.fill(): - * throw new BadRequestException({ error, intentId, minDstAmount, fillAmount }) - * — all four fields are intentionally public. - * - * To expose a new field from a custom-shaped exception, add its name here and - * document why it is safe to return to API consumers. Closes #304. - */ -const CUSTOM_BODY_ALLOWLIST = new Set([ - "error", // human-readable error message (required) - "intentId", // the intent that failed — already in the URL, safe to echo - "fillAmount", // the amount the solver attempted — safe to echo to the solver - "minDstAmount", // the required minimum — safe to echo to the solver - "requestId", // injected below; listed for clarity -]); - -interface JsonResponse { - status: (code: number) => { json: (body: unknown) => void }; - setHeader?: (name: string, value: string) => void; -} - -interface RequestWithId { - requestId?: string; -} - -/** - * Some 503s are transient and tell the client when to come back (an emergency - * pause, a shedding load-shed). An exception may expose `retryAfterSeconds` to - * have the `Retry-After` header set alongside the body. - */ -function setRetryAfter(response: JsonResponse, exception: unknown): void { - const retryAfter = (exception as { retryAfterSeconds?: unknown })?.retryAfterSeconds; - if (typeof retryAfter === "number" && Number.isFinite(retryAfter) && retryAfter > 0) { - response.setHeader?.("Retry-After", String(Math.ceil(retryAfter))); - } -} - -function addRequestId(body: Record, requestId?: string): Record { - if (requestId) body.requestId = requestId; - return body; -} - -@Catch() -export class HttpExceptionFilter implements ExceptionFilter { - catch(exception: unknown, host: ArgumentsHost) { - const response = host.switchToHttp().getResponse(); - const request = host.switchToHttp().getRequest(); - const requestId = request.requestId; - - if (exception instanceof HttpException) { - const status = exception.getStatus(); - const body = exception.getResponse(); - setRetryAfter(response, exception); - - if (typeof body === "string") { - response.status(status).json(addRequestId({ error: body }, requestId)); - return; - } - - if (typeof body === "object" && body !== null) { - const b = body as Record; - - if (Array.isArray(b.message)) { - response.status(status).json( - addRequestId({ error: "Validation failed", details: b.message }, requestId), - ); - return; - } - - // Custom-shaped bodies passed directly to an exception constructor, - // e.g. new BadRequestException({ error: "...", fillAmount, minDstAmount }) - // Only fields on CUSTOM_BODY_ALLOWLIST are forwarded to the client. - // Any extra field is stripped and a warning is emitted so that future - // contributors learn about the leak before it reaches production. - // Closes #304. - if (typeof b.error === "string" && !b.statusCode) { - const sanitized: Record = {}; - for (const [key, value] of Object.entries(b)) { - if (CUSTOM_BODY_ALLOWLIST.has(key)) { - sanitized[key] = value; - } else { - logger.warn( - `HttpExceptionFilter: custom exception body contains unexpected field "${key}" — ` + - "it has been stripped from the response. If this field is safe to expose to " + - "API consumers, add it to CUSTOM_BODY_ALLOWLIST in http-exception.filter.ts.", - ); - } - } - response.status(status).json(addRequestId(sanitized, requestId)); - return; - } - - if (typeof b.message === "string") { - response.status(status).json(addRequestId({ error: b.message }, requestId)); - return; - } - } - - response.status(status).json(addRequestId({ error: exception.message }, requestId)); - return; - } - - const err = exception instanceof Error ? exception : new Error("Unknown error"); - - // Express/body-parser errors (e.g. PayloadTooLargeError) carry a numeric - // `status` field. Propagate it as-is instead of masking with 500. - const httpStatus = (exception as Record)?.status; - if (typeof httpStatus === "number" && httpStatus >= 400 && httpStatus < 600) { - response.status(httpStatus).json({ error: err.message || "Request error" }); - return; - } - - logger.error(err.stack ?? err.message); - // Alert on-call engineers — only fires for unexpected exceptions, not - // routine HttpExceptions, so alert fatigue on 404 / 400 is avoided. - captureException(err); - response.status(500).json({ error: err.message || "Internal server error" }); - } -} diff --git a/src/config/env.validation.spec.ts b/src/config/env.validation.spec.ts index f0102c3a..64073e8f 100644 --- a/src/config/env.validation.spec.ts +++ b/src/config/env.validation.spec.ts @@ -11,12 +11,15 @@ const VALID_KEY = "S" + "A".repeat(55); * ONCHAIN_DRY_RUN (#260), SOROBAN_SIGNING_KEY, and KILLSWITCH_OPERATOR_TOKEN * (#477). Each test below overrides only the one key it is about, so a failure * is attributable to that key rather than to whichever requirement fired first. + * + * METRICS_TOKEN (#298) is also required in production. */ const PROD_ENV = { NODE_ENV: "production", ONCHAIN_DRY_RUN: true, SOROBAN_SIGNING_KEY: VALID_KEY, KILLSWITCH_OPERATOR_TOKEN: "operator-secret", + METRICS_TOKEN: "a-sufficiently-long-metrics-secret", }; describe("envValidationSchema — SOROBAN_SIGNING_KEY", () => { @@ -259,3 +262,64 @@ describe("envValidationSchema — kill-switch propagation (issue #477)", () => { expect(value.KILLSWITCH_REDIS_URL).toBe(""); }); }); + +describe("envValidationSchema — METRICS_TOKEN (issue #298)", () => { + it("defaults to an empty string outside production (endpoint disabled)", () => { + const { error, value } = envValidationSchema.validate(BASE_ENV); + expect(error).toBeUndefined(); + expect(value.METRICS_TOKEN).toBe(""); + }); + + it("accepts an explicit empty value outside production", () => { + const { error } = envValidationSchema.validate({ + ...BASE_ENV, + METRICS_TOKEN: "", + }); + expect(error).toBeUndefined(); + }); + + it("accepts any non-empty token outside production", () => { + const { error, value } = envValidationSchema.validate({ + ...BASE_ENV, + METRICS_TOKEN: "short", + }); + expect(error).toBeUndefined(); + expect(value.METRICS_TOKEN).toBe("short"); + }); + + it("is required (non-empty) in production", () => { + const { error } = envValidationSchema.validate({ + ...PROD_ENV, + METRICS_TOKEN: undefined, + }); + expect(error).toBeDefined(); + expect(error?.message).toContain("METRICS_TOKEN"); + }); + + it("rejects an empty string in production", () => { + const { error } = envValidationSchema.validate({ + ...PROD_ENV, + METRICS_TOKEN: "", + }); + expect(error).toBeDefined(); + expect(error?.message).toContain("METRICS_TOKEN"); + }); + + it("rejects a token shorter than 16 chars in production", () => { + const { error } = envValidationSchema.validate({ + ...PROD_ENV, + METRICS_TOKEN: "too-short", + }); + expect(error).toBeDefined(); + expect(error?.message).toContain("METRICS_TOKEN"); + }); + + it("accepts a token of at least 16 characters in production", () => { + const { error, value } = envValidationSchema.validate({ + ...PROD_ENV, + METRICS_TOKEN: "a-sufficiently-long-metrics-secret", + }); + expect(error).toBeUndefined(); + expect(value.METRICS_TOKEN).toBe("a-sufficiently-long-metrics-secret"); + }); +}); diff --git a/src/metrics/metrics.controller.ts b/src/metrics/metrics.controller.ts index 14763583..e69de29b 100644 --- a/src/metrics/metrics.controller.ts +++ b/src/metrics/metrics.controller.ts @@ -1,32 +0,0 @@ -import { Controller, Get, Header, Inject, UseGuards } from "@nestjs/common"; -import { ApiTags } from "@nestjs/swagger"; -import { MetricsService } from "./metrics.service"; -import { MetricsTokenGuard } from "./metrics-token.guard"; - -@ApiTags("metrics") -@Controller("metrics") -export class MetricsController { - constructor(@Inject(MetricsService) private readonly metricsService: MetricsService) {} - - /** - * Prometheus scrape endpoint. - * - * Access is controlled by MetricsTokenGuard: - * - With METRICS_TOKEN set: require `Authorization: Bearer `. - * - Without METRICS_TOKEN in non-production: open (local dev / test). - * - Without METRICS_TOKEN in production: always 401 (fail closed). - * - * Configure your Prometheus scrape job with: - * authorization: - * type: Bearer - * credentials: - * - * Closes #298. - */ - @Get() - @UseGuards(MetricsTokenGuard) - @Header("Content-Type", "text/plain; charset=utf-8") - async index(): Promise { - return this.metricsService.metrics(); - } -} diff --git a/test/metrics.e2e-spec.ts b/test/metrics.e2e-spec.ts index 8d0225fe..d6848156 100644 --- a/test/metrics.e2e-spec.ts +++ b/test/metrics.e2e-spec.ts @@ -2,30 +2,62 @@ import { INestApplication } from "@nestjs/common"; import request from "supertest"; import { createTestApp } from "./utils/create-test-app"; +const TEST_METRICS_TOKEN = "test-metrics-token-for-ci"; + describe("MetricsController (e2e)", () => { let app: INestApplication; beforeAll(async () => { + process.env.METRICS_TOKEN = TEST_METRICS_TOKEN; app = await createTestApp(); }); afterAll(async () => { + delete process.env.METRICS_TOKEN; await app.close(); }); - it("GET /metrics returns prometheus metrics", async () => { - const res = await request(app.getHttpServer()).get("/metrics").expect(200); + describe("authentication", () => { + it("GET /metrics returns 403 with no Authorization header", async () => { + await request(app.getHttpServer()).get("/metrics").expect(403); + }); + + it("GET /metrics returns 403 with wrong token", async () => { + await request(app.getHttpServer()) + .get("/metrics") + .set("Authorization", "Bearer wrong-token") + .expect(403); + }); - expect(res.headers["content-type"]).toContain("text/plain"); - expect(res.text).toContain("vortex_http_requests_total"); - expect(res.text).toContain("vortex_http_request_duration_seconds"); - expect(res.text).toContain("vortex_http_request_errors_total"); - expect(res.text).toContain("vortex_intent_state_transitions_total"); - expect(res.text).toContain("vortex_ws_connections_active"); + it("GET /metrics returns 403 when Authorization header has no Bearer prefix", async () => { + await request(app.getHttpServer()) + .get("/metrics") + .set("Authorization", TEST_METRICS_TOKEN) + .expect(403); + }); }); - it("GET /metrics includes default metrics (process_cpu)", async () => { - const res = await request(app.getHttpServer()).get("/metrics").expect(200); - expect(res.text).toContain("vortex_process_cpu_seconds"); + describe("authorised access", () => { + it("GET /metrics returns prometheus metrics with valid token", async () => { + const res = await request(app.getHttpServer()) + .get("/metrics") + .set("Authorization", `Bearer ${TEST_METRICS_TOKEN}`) + .expect(200); + + expect(res.headers["content-type"]).toContain("text/plain"); + expect(res.text).toContain("vortex_http_requests_total"); + expect(res.text).toContain("vortex_http_request_duration_seconds"); + expect(res.text).toContain("vortex_http_request_errors_total"); + expect(res.text).toContain("vortex_intent_state_transitions_total"); + expect(res.text).toContain("vortex_ws_connections_active"); + }); + + it("GET /metrics includes default metrics (process_cpu)", async () => { + const res = await request(app.getHttpServer()) + .get("/metrics") + .set("Authorization", `Bearer ${TEST_METRICS_TOKEN}`) + .expect(200); + expect(res.text).toContain("vortex_process_cpu_seconds"); + }); }); });