From 16747f6db29e28bb143c69a1884a9aa13bdfe3b7 Mon Sep 17 00:00:00 2001 From: "flamingo[bot]" <277372822+flamingo[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 18:05:13 +0000 Subject: [PATCH 1/2] =?UTF-8?q?chore:=20add=20=F0=9F=A6=A9=20Flamingo=20Co?= =?UTF-8?q?de=20Documentation=20workflow?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit This workflow enables automated documentation generation using the ๐Ÿฆฉ Flamingo Code Documentation pipeline. It runs on repository_dispatch events triggered by the multi-platform-hub. No secrets required - all credentials are passed securely at runtime. --- .github/workflows/code-documentation.yml | 3246 ++++++++++++++++++++++ 1 file changed, 3246 insertions(+) create mode 100644 .github/workflows/code-documentation.yml diff --git a/.github/workflows/code-documentation.yml b/.github/workflows/code-documentation.yml new file mode 100644 index 00000000..3369410f --- /dev/null +++ b/.github/workflows/code-documentation.yml @@ -0,0 +1,3246 @@ +# Flamingo Code Documentation +# ============================================================================= +# Installed into a target repository by the multi-platform hub ("Setup +# Workflow"). The hub dispatches it to generate this repository's +# documentation and open a pull request with the result. +# +# Jobs +# code-graph Deterministic code graph (no model call). Runs on pushes to +# the default branch, on the hub's `flamingo-code-graph` +# re-dispatch, and on a manual run with graph_only=true. +# doc-pipeline The documentation run, dispatched by the hub: +# Stage 0 code graph (same build as the code-graph job) +# Stage 1 inline docs, one hidden .md beside each source file +# Stage 2 reference docs: CodeWiki, or the Claude +# architecture analysis where CodeWiki cannot parse +# the primary language +# Stage 3 tutorials (getting started, development) +# Stage 4 repository docs: README, CONTRIBUTING, docs index +# +# Stages 2 (Claude), 3 and 4 follow the code reviewer's agentic pattern +# (templates/scripts/code-documentation-lib.mjs): the hub serves this +# repository's settings, the script packs the material, ONE hub call writes +# ONE document (forced `emit_document`; the hub's read tools when the +# repository's "Graph lookups" switch is on), a deterministic gate checks every +# repository path it names, and the script, never the model, picks where each +# document is written. +# +# Fleet contracts (renaming any of these needs a fleet-wide re-push): the +# installed path .github/workflows/code-documentation.yml (the setup PR +# hard-deletes the old doc-orchestrator.yml), the repository_dispatch type +# `code-documentation`, the secrets below, and the script file names the hub serves. +# +# Repository secrets (Settings > Secrets and variables > Actions) +# FLAMINGO_HUB_SECRET Required. Authenticates every hub call. +# ANTHROPIC_API_KEY CodeWiki (stage 2) only; every other stage calls +# Claude through the hub. +# OPENAI_API_KEY CodeWiki (stage 2). +# YOUTUBE_API_KEY Optional. YouTube embeds in stages 3 and 4. + +name: ๐Ÿฆฉ Flamingo Code Documentation + +on: + # Push trigger โ€” two things ride it. It registers the workflow with GitHub + # Actions (required for the workflow_dispatch API), and on the repository's + # DEFAULT branch it runs the `code-graph` job below, which re-indexes the + # code graph the hub serves to the code reviewer and to the documentation + # stages. The documentation pipeline itself NEVER runs on push (see its + # `if:`). Documentation and markdown are ignored on purpose: a docs PR + # merging must not rebuild a graph that only source files can change. + push: + paths-ignore: + - 'docs/**' + - '**.md' + + # Multi-repo change sets, LIVE: every pull request event (a `Depends-On:` line + # added, a push, a merge) asks the hub to refresh its change sets at once, and + # the same run builds the PR's overlay when the hub says it is a member whose + # overlay is missing or stale. ONLY the code-graph job runs on this event (see + # both jobs' `if:`); no path filter, because a description edit is the event. + pull_request: + types: [opened, reopened, edited, synchronize, ready_for_review, closed] + + repository_dispatch: + # `code-documentation` (CODE_DOCUMENTATION_DISPATCH_EVENT_TYPE) runs the + # documentation pipeline. `flamingo-code-graph` + # (CODE_GRAPH_DISPATCH_EVENT_TYPE in lib/config/code-graph-workflow.ts) + # runs ONLY the graph job; the hub's reconcile job sends it when a + # repository's graph is missing or stale. + types: [code-documentation, flamingo-code-graph] + + workflow_dispatch: + # โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ• + # GENERATED FROM SINGLE SOURCE OF TRUTH: lib/config/code-documentation-params.ts + # This section is auto-generated at runtime when creating workflow PRs + # โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ• + inputs: + # Individual parameters + run_id: + description: 'Unique execution ID' + required: true + repo_id: + description: 'Repository ID from database' + required: true + hub_base_url: + description: 'Hub base URL (e.g., https://product-hub.flamingo.so)' + required: true + stages: + description: 'Pipeline stages to execute' + required: true + dependencies: + description: 'Comma-separated dependency repos' + required: false + default: '' + source_branch: + description: 'Branch to analyze code from' + required: true + source_files_limit: + description: 'Max source files to process (0 = unlimited)' + required: true + claude_model: + description: 'Claude model ID for Stage 1/3/4 + Stage 2 Claude-arch fallback (SSOT from hub)' + required: true + codewiki_config: + description: 'Complete CodeWiki configuration (per-phase models, engine, stages, depth)' + required: true + output_paths: + description: 'Output paths configuration' + required: true + timeout: + description: 'Timeout in hours' + required: true + youtube_config: + description: 'YouTube integration configuration (channels only - API key in secrets)' + required: true + readme_config: + description: 'README logo configuration' + required: true + custom_repo_instructions: + description: 'Custom AI instructions' + required: false + default: '' + external_repos: + description: 'External repos JSON' + required: false + default: '[]' + stage_count: + description: 'Total number of pipeline stages' + required: false + default: '4' + graph_only: + description: 'true = run only the code-graph job (no documentation run)' + required: false + default: 'false' + # โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ• + # END GENERATED SECTION + # โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ• + +env: + # SECURITY: Only NON-SENSITIVE variables in job-level env + # Secrets are passed per-step to avoid exposure in job setup logs + # Run configuration (non-sensitive) + RUN_ID: ${{ github.event.client_payload.run_id || github.event.inputs.run_id || github.run_id }} + REPO_ID: ${{ github.event.client_payload.repo_id || github.event.inputs.repo_id || '' }} + # A push event carries no payload, so the graph job falls back to the org + # Actions variable โ€” the same fallback the code-review workflow uses. + HUB_BASE_URL: ${{ github.event.client_payload.hub_base_url || github.event.inputs.hub_base_url || vars.FLAMINGO_HUB_BASE_URL || '' }} + # 'true' = run only the code-graph job (a manual workflow_dispatch; the hub's + # own documentation dispatches always send 'false'). + GRAPH_ONLY: ${{ github.event.client_payload.graph_only || github.event.inputs.graph_only || 'false' }} + # No literal fallback: the stage list lives in CODE_DOCUMENTATION_STAGES on the + # hub and is lifted into the payload per repo. A literal here would silently + # restore all four stages on a payload gap, and "Remove docs this run + # regenerates" would already have wiped the docs tree: the run would delete + # docs and regenerate nothing. + STAGES: ${{ github.event.client_payload.stages || github.event.inputs.stages || '' }} + DEPENDENCIES: ${{ github.event.client_payload.dependencies || github.event.inputs.dependencies || '' }} + STAGE_COUNT: ${{ github.event.client_payload.stage_count || github.event.inputs.stage_count || '4' }} + # Branch to checkout for code analysis (github_branch from repo config) + SOURCE_BRANCH: ${{ github.event.client_payload.source_branch || github.event.inputs.source_branch || 'main' }} + # Debug/testing: limit total source files to analyze (0=unlimited) + # Files beyond this limit are DELETED - all stages then process remaining files + SOURCE_FILES_LIMIT: ${{ github.event.client_payload.source_files_limit || github.event.inputs.source_files_limit || '0' }} + # Claude model SSOT: the hub's CODE_DOCUMENTATION_DEFAULT_MODEL + # (lib/constants/ai-models.ts). The stage 1, 3 and 4 generators and the + # stage 2 Claude analysis all read this variable; no shipped script holds a + # literal model id, and every one throws if CLAUDE_MODEL is empty. Re-run + # "Setup Workflow" after bumping the hub-side constant to propagate it. + # There is deliberately NO companion request-shape variable: every stage + # except CodeWiki calls Claude through the hub (/api/ci/claude), which + # resolves the model's request shape itself. + CLAUDE_MODEL: ${{ github.event.client_payload.claude_model || github.event.inputs.claude_model || '' }} + # ============================================================================= + # JSON-grouped parameters to stay under GitHub Actions 25-parameter limit + # These are parsed early in the workflow to extract individual values + # ============================================================================= + # NOTE: these fall back to EMPTY, not '{}'. An empty-object default made the + # `[ -z ... ]` presence checks below unreachable, so a missing payload silently + # produced `null` for every jq lookup and propagated as `--cluster-model null`. + # The hub always sends a complete, deep-merged blob (buildPayloadFromRepo). + CODEWIKI_CONFIG_JSON: ${{ github.event.client_payload.codewiki_config || github.event.inputs.codewiki_config || '' }} + OUTPUT_PATHS_JSON: ${{ github.event.client_payload.output_paths || github.event.inputs.output_paths || '' }} + # Stage timeouts (in hours) โ€” single source of truth, downstream steps reference + # `env.STAGE_TIMEOUT_HOURS` directly. Stage 4 is the exception (1h vs 24h cap). + STAGE_TIMEOUT_HOURS: ${{ github.event.client_payload.timeout || github.event.inputs.timeout || '24' }} + # Aliases preserved for downstream step env: keys (they reference these names + # by string). All resolve to the same single source. + STAGE1_TIMEOUT_HOURS: ${{ github.event.client_payload.timeout || github.event.inputs.timeout || '24' }} + # Stage 1 incremental push: commit + push the PR branch every N generated inline + # docs. A 24h stage that gets cancelled used to lose ALL of its work because the + # only commit happened after the generator returned. Bound the loss to N files. + STAGE1_PUSH_INTERVAL: ${{ github.event.client_payload.stage1_push_interval || '100' }} + STAGE2_TIMEOUT_HOURS: ${{ github.event.client_payload.timeout || github.event.inputs.timeout || '24' }} + STAGE3_TIMEOUT_HOURS: ${{ github.event.client_payload.timeout || github.event.inputs.timeout || '24' }} + # Stage 4: Repository Documentation + TEMPLATE_REPO: ${{ github.event.client_payload.template_repo || 'flamingo-stack/openframe-oss-tenant' }} + TEMPLATE_BRANCH: ${{ github.event.client_payload.template_branch || 'main' }} + STAGE4_TIMEOUT_HOURS: ${{ github.event.client_payload.stage4_timeout || '1' }} + # YouTube Integration (Stage 3 + Stage 4) - JSONB configuration (API key from secrets) + YOUTUBE_CONFIG_JSON: ${{ github.event.client_payload.youtube_config || github.event.inputs.youtube_config || '' }} + # README Configuration - JSONB configuration for logo branding + README_CONFIG_JSON: ${{ github.event.client_payload.readme_config || github.event.inputs.readme_config || '' }} + # Custom AI Instructions (All Stages) - Repository-specific instructions for AI generation + CUSTOM_REPO_INSTRUCTIONS: ${{ github.event.client_payload.custom_repo_instructions || github.event.inputs.custom_repo_instructions || '' }} + # External Repositories - JSON array of external repo configurations + EXTERNAL_REPOS: ${{ github.event.client_payload.external_repos || github.event.inputs.external_repos || '[]' }} + # ======================================================================== + # Repository Context - CRITICAL for preventing AI URL hallucinations + # These values are passed to ALL AI prompts to ensure correct GitHub URLs + # ======================================================================== + GITHUB_REPOSITORY: ${{ github.repository }} # e.g., "flamingo-stack/openframe-oss-tenant" + GITHUB_REPOSITORY_OWNER: ${{ github.repository_owner }} # e.g., "flamingo-stack" + GITHUB_SERVER_URL: ${{ github.server_url }} # e.g., "https://github.com" + # Analysis Exclusions - Complete array of glob patterns to exclude from repository analysis + # (build artifacts, dependencies, the hub's own checkout, cloned dependency repos) + EXCLUDED_PATHS: '**/node_modules/**,**/.git/**,**/target/**,**/dist/**,**/build/**,**/.next/**,**/out/**,**/coverage/**,**/vendor/**,**/.yalc/**,**/.turbo/**,**/.gradle/**,**/__pycache__/**,**/.terraform/**,**/.venv/**,**/venv/**,**/multi-platform-hub/**,**/deps-*/**' + README_LOGO_ALT: 'OpenFrame Logo' + +jobs: + # =========================================================================== + # CODE GRAPH: deterministic, no model call. Tags every public symbol, + # import and manifest of the checkout (code-graph-build.mjs) and uploads the + # result to the hub, which promotes a default-branch snapshot to `live` and + # serves it to the code reviewer (consumers of a symbol a PR removes) and to + # the documentation stages (the derived ecosystem.md). Runs on every push to + # the default branch, on the hub's `flamingo-code-graph` re-dispatch, and on + # a manual workflow_dispatch with graph_only=true. There is no webhook + # callback: the upload IS the report. The same build also runs as stage 0 of + # a full documentation run (inside doc-pipeline, below). + # =========================================================================== + code-graph: + runs-on: ubuntu-latest + timeout-minutes: 20 + permissions: + contents: read + # THREE groups, so no two kinds of run ever cancel each other: + # - a `pull_request` event runs in `-pr-`, and NEVER cancels a running one + # (`cancel-in-progress` is false for it): a newer event QUEUES behind the run + # in progress (GitHub keeps one pending run per group, replacing an older + # pending one, so the newest event still runs). Cancelling would kill an + # overlay build the hub already recorded as this PR's own (a description + # edit or the merge seconds after a push), and nothing would rebuild that + # head until the retry window passed; + # - an OVERLAY build the hub's `code-change-sets` job dispatches + # (overlay_pr / overlay_head_sha in the payload) runs in `-overlay-`: a + # newer dispatch for the same PR supersedes the older one; + # - everything else (the default branch's rebuild) keeps the repository group. + # A dispatched build and the PR's own run may build the same head at once; + # that is harmless: the hub upserts an overlay on (repo, pr, head, generator). + # A bot's own description edit (the hub writing its block) is excluded by the + # `if:` below, but a job may join its concurrency group before that `if:` is + # evaluated: it gets a group of its own (`-bot-`), so it can never + # cancel the PR's run that is building the overlay the same refresh asked for. + concurrency: + group: flamingo-code-graph-${{ github.repository }}${{ github.event.client_payload.overlay_pr && format('-overlay-{0}', github.event.client_payload.overlay_pr) || github.event.pull_request.number && format('-pr-{0}', github.event.pull_request.number) || '' }}${{ (github.event.action == 'edited' && github.event.sender.type == 'Bot') && format('-bot-{0}', github.run_id) || '' }} + cancel-in-progress: ${{ github.event_name != 'pull_request' }} + if: >- + (github.event_name == 'push' && github.ref == format('refs/heads/{0}', github.event.repository.default_branch)) + || github.event.action == 'flamingo-code-graph' + || (github.event_name == 'workflow_dispatch' && github.event.inputs.graph_only == 'true') + || (github.event_name == 'pull_request' + && github.event.pull_request.head.repo.full_name == github.repository + && !startsWith(github.event.pull_request.head.ref, 'ai-fix/') + && !startsWith(github.event.pull_request.head.ref, 'setup/') + && !(github.event.action == 'edited' && github.event.sender.type == 'Bot')) + # The overlay coordinates, spelled ONCE: the hub's dispatch payload, or this PR's own event (the steps + # below build only after the live refresh answered `build`). Both empty on a default-branch rebuild. + env: + OVERLAY_PR: ${{ github.event.client_payload.overlay_pr || github.event.pull_request.number || '' }} + OVERLAY_HEAD_SHA: ${{ github.event.client_payload.overlay_head_sha || github.event.pull_request.head.sha || '' }} + + steps: + # Fail LOUD, not silent: a push on a repo whose org never set + # FLAMINGO_HUB_BASE_URL would otherwise curl an empty origin and die with + # an unrelated error. Also normalizes a trailing slash ONCE. + - name: Validate configuration + id: config + # A pull_request event must never turn a PR check red: the refresh is advisory + # (the hub's webhook path and schedule are the net), so a missing hub URL or + # secret on a PR only skips it. + continue-on-error: ${{ github.event_name == 'pull_request' }} + run: | + if [ -z "$HUB_BASE_URL" ]; then + echo "::error title=Hub URL missing::HUB_BASE_URL is empty. Set the organization Actions variable FLAMINGO_HUB_BASE_URL, or pass hub_base_url in the dispatch payload." + exit 1 + fi + echo "HUB_BASE_URL=${HUB_BASE_URL%/}" >> "$GITHUB_ENV" + echo "Hub: ${HUB_BASE_URL%/}" + + # The shared script bootstrap (byte-mirrored from workflow-scripts-bootstrap.ts, + # asserted by the build gate). BOTH graph scripts are downloaded: the builder + # imports ./code-graph-lib.mjs from its own directory. + - name: Download graph scripts + id: scripts + if: github.event_name != 'pull_request' || steps.config.outcome == 'success' + continue-on-error: ${{ github.event_name == 'pull_request' }} + env: + WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }} + run: | + + # Function to download and verify script + SCRIPT_MANIFEST=/tmp/flamingo-script-manifest.json + # WEBHOOK_SECRET reaches curl through a 0600 config file, never argv โ€” see + # curlAuthPreamble, which always traps the removal. + CURL_CFG=$(mktemp) && chmod 600 "$CURL_CFG" + trap 'rm -f "$CURL_CFG"' EXIT + printf 'header = "Authorization: Bearer %s"\n' "$WEBHOOK_SECRET" > "$CURL_CFG" + # The scripts surface. load_script_manifest pins SCRIPTS_BASE_URL to it. + CI_SCRIPTS_URL="${HUB_BASE_URL%/}/api/ci/scripts" + + # _try_manifest โ€” 0 loaded, 1 no manifest surface there, 2 fatal. + # The manifest is asked for ONE group: its keys are the files to download. + _try_manifest() { + local base="$1" code + code=$(curl -sS -w '%{http_code}' -o "$SCRIPT_MANIFEST" \ + -K "$CURL_CFG" \ + "$base/manifest.json?group=$SCRIPT_GROUP") || code="000" + + if [ "$code" = "404" ]; then rm -f "$SCRIPT_MANIFEST"; return 1; fi + if [ "$code" != "200" ]; then + echo "โŒ manifest request to $base failed (HTTP $code)" + rm -f "$SCRIPT_MANIFEST" + return 2 + fi + # The digests are the TOP-LEVEL object. successResponse is the standard + # emitter but it does NOT add a wrapper โ€” it is NextResponse.json(data) + # plus the no-store header โ€” so there is no .data to reach through. + # A 200 that is not a manifest is how a hub which does not serve this path + # answers (the proxy rewrites unknown routes and returns HTML), so it + # means "wrong surface", not "corrupt". + if ! jq -e 'type == "object" and length > 0 and (to_entries | all(.value | type == "string"))' "$SCRIPT_MANIFEST" >/dev/null 2>&1; then + rm -f "$SCRIPT_MANIFEST" + return 1 + fi + return 0 + } + + # load_script_manifest + load_script_manifest() { + SCRIPT_GROUP="$1" + # "cmd; rc=$?" dies under the set -euo pipefail these steps run with โ€” + # errexit fires before rc is read and the step ends with NO output. And + # "if ! cmd; then rc=$?" is worse: inside the branch $? is the status of + # the NEGATION (0), so every failure reads as success. "|| rc=$?" is the + # one form that both suppresses errexit and preserves the real code. + local rc=0 + _try_manifest "$CI_SCRIPTS_URL" || rc=$? + if [ "$rc" = "0" ]; then + SCRIPTS_BASE_URL="$CI_SCRIPTS_URL" + echo "โœ… script manifest loaded ($(jq -r 'length' "$SCRIPT_MANIFEST") scripts)" + return 0 + fi + if [ "$rc" = "2" ]; then exit 1; fi + + # No manifest on the scripts surface: a hub older than the manifest + # itself. The manifest is the file list, so there is nothing to download. + echo "โŒ no script manifest on this hub ($HUB_BASE_URL): it cannot name the $SCRIPT_GROUP scripts. Redeploy the hub." + exit 1 + } + + # download_script_group โ€” the hub names the files, this workflow + # names only the group. Downloads every script of the group, in served order. + download_script_group() { + load_script_manifest "$1" + local name + # The loop runs in THIS shell (no pipe), so a failed download exits the step. + while IFS= read -r name; do + download_and_verify "$name" + done < <(jq -r 'keys_unsorted[]' "$SCRIPT_MANIFEST") + } + + download_and_verify() { + local script_name="$1" + local output_path="/tmp/$script_name" + + local expected_hash + expected_hash=$(jq -r --arg n "$script_name" '.[$n] // empty' "$SCRIPT_MANIFEST") + if [ -z "$expected_hash" ]; then + echo "โŒ $script_name is not in the server's script manifest!" + echo " The hub serves no such script, or it failed to read on the server." + exit 1 + fi + if ! printf '%s' "$expected_hash" | grep -Eq '^[0-9a-f]{64}$'; then + echo "โŒ the manifest entry for $script_name is not a SHA-256 digest โ€” refusing to run it." + exit 1 + fi + + curl -fsSL "$SCRIPTS_BASE_URL/$script_name" \ + -K "$CURL_CFG" \ + -o "$output_path" + + local actual_hash=$(shasum -a 256 "$output_path" | cut -d' ' -f1) + + if [ "$actual_hash" != "$expected_hash" ]; then + echo "โŒ HASH MISMATCH for $script_name!" + echo " Expected: $expected_hash" + echo " Actual: $actual_hash" + echo " The download was corrupted in transit โ€” both values come from the same deployment." + exit 1 + fi + + # Make shell scripts executable + if [[ "$script_name" == *.sh ]]; then + chmod +x "$output_path" + fi + + echo "โœ… $script_name verified (hash: ${actual_hash:0:16}...)" + } + + # Digests AND the file list come from the deployment serving the bytes, + # not from this file: the step names a group (SCRIPT_GROUPS in the hub's + # lib/config/ci-script-catalog.ts) and downloads what the hub lists for it. + download_script_group "code-graph" + + # LIVE change sets (a `pull_request` event): ask the hub to refresh NOW and + # whether this run builds the PR's overlay (`build`). Never fails the run. + # A fork has no secrets and is excluded by the job's `if:`, as are the hub + # tools' own branches (TOOL_BRANCH_PREFIXES); a bot's own description edit + # (the hub writing its block) is excluded too, or every hub write would + # start another run. + - name: Refresh the change set + id: refresh + if: github.event_name == 'pull_request' && steps.scripts.outcome == 'success' + # A hub that cannot answer never fails the pull request's check: the schedule links the set. + continue-on-error: true + env: + WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }} + GITHUB_REPOSITORY: ${{ github.repository }} + CODE_GRAPH_REFRESH_PR: ${{ github.event.pull_request.number }} + # The head of THIS event: the hub answers `build` only while it is still the PR's live head. + CODE_GRAPH_REFRESH_HEAD_SHA: ${{ github.event.pull_request.head.sha }} + # The event's action: the hub runs its job only for one that can change a set (CHANGE_SET_REFRESH_ACTIONS). + CODE_GRAPH_REFRESH_ACTION: ${{ github.event.action }} + # The event's sender type: the hub refuses a bot's own description edit too, not only this job's `if:`. + CODE_GRAPH_REFRESH_SENDER_TYPE: ${{ github.event.sender.type }} + run: node /tmp/code-graph-build.mjs + + # FULL history, blobless. `collectFileFacts` derives per-file ownership + # (last commit, recent commits, commits in the window) from one + # `git log --no-merges --no-renames` walk, which a depth-1 checkout + # cannot answer โ€” it would report every file as owned by one commit. + # `filter: blob:none` keeps the clone cheap: the walk reads commit + # metadata and name-only paths, never file contents, which is also why + # the walk passes `--no-renames` (rename detection would fetch blobs). + # A shallow checkout still degrades gracefully: ownership is omitted and + # `coverage.ownership` is false rather than the job failing. + # In overlay mode the checkout is the PR HEAD, with every blob: the + # overlay diffs it against the live snapshot commit with rename + # detection, which reads contents a blob:none clone cannot fetch here. + # + # The four build steps below share one `if:` (a PR event builds only when the refresh answered + # `build`) and one `continue-on-error` (a pull request's check never goes red on the overlay: the + # graph lane and the schedule rebuild it). Actions has no step group, so each step states both. + - name: Check out repository + if: github.event_name != 'pull_request' || steps.refresh.outputs.build == 'true' + continue-on-error: ${{ github.event_name == 'pull_request' }} + uses: actions/checkout@fbc6f3992d24b796d5a048ff273f7fcc4a7b6c09 # v5.1.0 + with: + ref: ${{ env.OVERLAY_HEAD_SHA }} + fetch-depth: 0 + # `!x && 'blob:none' || ''` โ€” never `x && '' || โ€ฆ`: '' is falsy in an expression, so that form always yields blob:none. + filter: ${{ !env.OVERLAY_PR && 'blob:none' || '' }} + persist-credentials: false + + - name: Set up Node.js + if: github.event_name != 'pull_request' || steps.refresh.outputs.build == 'true' + continue-on-error: ${{ github.event_name == 'pull_request' }} + uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0 + with: + node-version: '22' + + # The command is CODE_GRAPH_INSTALL_COMMAND (lib/config/code-graph-workflow.ts), + # the one spelling both workflows use: pinned wasm tree-sitter + grammars + + # yaml into an isolated tree under RUNNER_TEMP, exported as CODE_GRAPH_DEPS_DIR. + - name: Install graph dependencies + if: github.event_name != 'pull_request' || steps.refresh.outputs.build == 'true' + continue-on-error: ${{ github.event_name == 'pull_request' }} + run: mkdir -p "$RUNNER_TEMP/code-graph-deps" && cd "$RUNNER_TEMP/code-graph-deps" && printf '%s' '{"name":"code-graph-deps","version":"1.0.0","private":true,"dependencies":{"web-tree-sitter":"0.27.0","@vscode/tree-sitter-wasm":"0.3.1","yaml":"2.9.1"}}' > package.json && printf '%s' '{"name":"code-graph-deps","version":"1.0.0","lockfileVersion":3,"requires":true,"packages":{"":{"name":"code-graph-deps","version":"1.0.0","dependencies":{"web-tree-sitter":"0.27.0","@vscode/tree-sitter-wasm":"0.3.1","yaml":"2.9.1"}},"node_modules/web-tree-sitter":{"version":"0.27.0","resolved":"https://registry.npmjs.org/web-tree-sitter/-/web-tree-sitter-0.27.0.tgz","integrity":"sha512-XK08gj6RwTMQatAG7uVRP8MunqotL/XC19vHgkSPKmELgbGPBj4ECvB8haHOUnyj6ls2B8t42UTro14zxGgAHg=="},"node_modules/@vscode/tree-sitter-wasm":{"version":"0.3.1","resolved":"https://registry.npmjs.org/@vscode/tree-sitter-wasm/-/tree-sitter-wasm-0.3.1.tgz","integrity":"sha512-RJFoomET6FajjG511fmQxeBQfU6M24a0aFZPqpid+ttIxanWf1VGytBG0UmsGjt07qmIPJS8U31D+aecuCucsQ=="},"node_modules/yaml":{"version":"2.9.1","resolved":"https://registry.npmjs.org/yaml/-/yaml-2.9.1.tgz","integrity":"sha512-3NxN8+78OdzbT7C/WjGsyfPAtJaN3FNDsWxv7Y7mcDsT/oOmgW8BpyQQFFBnvZE3j9Y2Sdz1ULFLezL7Eb2yFw=="}}}' > package-lock.json && npm ci --ignore-scripts --no-audit --no-fund && echo "CODE_GRAPH_DEPS_DIR=$RUNNER_TEMP/code-graph-deps" >> "$GITHUB_ENV" || { echo "::warning::graph dependencies failed their lockfile-enforced install; continuing without them"; rm -rf "$RUNNER_TEMP/code-graph-deps"; exit 1; } + + # Overlay mode posts the PR's overlay INSTEAD of a snapshot (code-graph-build.mjs `overlayMain`). + - name: Build and upload the code graph + id: graph + if: github.event_name != 'pull_request' || steps.refresh.outputs.build == 'true' + continue-on-error: ${{ github.event_name == 'pull_request' }} + env: + WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }} + CODE_GRAPH_DEPS_DIR: ${{ env.CODE_GRAPH_DEPS_DIR }} + GITHUB_REPOSITORY: ${{ github.repository }} + # Overlay mode: the job's overlay coordinates (empty outside it). + CODE_GRAPH_OVERLAY_PR: ${{ env.OVERLAY_PR }} + CODE_GRAPH_OVERLAY_HEAD_SHA: ${{ env.OVERLAY_HEAD_SHA }} + # An overlay run names a NON-default branch, so a hub or builder that predates overlay mode lands + # the PR head as a `branch` snapshot, never as the repository's live graph (the e2e run found an + # older builder promoting an unmerged PR's head live). Empty outside overlay mode. + CODE_GRAPH_BRANCH: ${{ env.OVERLAY_PR && format('overlay/pr-{0}', env.OVERLAY_PR) || '' }} + run: node /tmp/code-graph-build.mjs + + # =========================================================================== + # DOCUMENTATION PIPELINE: stages 0 to 4, one pull request per run. Every + # stage commits and pushes its own output as it finishes, so a timeout or a + # cancellation loses at most the stage in flight. + # =========================================================================== + doc-pipeline: + runs-on: ubuntu-latest + timeout-minutes: 720 # 12 hours for large repositories with many files + # contents: push the docs branch. pull-requests: open and update the pull + # request. issues: `gh label create` for the documentation / automated / + # in-progress labels; without it a repository that lacks them answers 403 + # on the label and then 422 on `gh pr create --label`. + permissions: + contents: write + pull-requests: write + issues: write + # Never on push (a push registers the workflow and runs the code-graph job + # only), never on the graph-only re-dispatch, never on a graph-only manual + # run. This is what lets workflow_dispatch API calls work on feature branches. + if: github.event_name != 'push' && github.event_name != 'pull_request' && github.event.action != 'flamingo-code-graph' && github.event.inputs.graph_only != 'true' + + steps: + # ========================================================================= + # REPORT CAPABILITY FIRST (shared failure-net standard with the code-review + # workflow): workflow-helpers.sh โ€” which carries send_webhook and the + # report/stage-callback helpers โ€” downloads in its OWN step before anything + # else, so a failure in the main script download below can still be pinged + # and reported home instead of leaving a phantom pending/running row. + # ========================================================================= + - name: Download report helpers + env: + WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }} + run: | + + # Function to download and verify script + SCRIPT_MANIFEST=/tmp/flamingo-script-manifest.json + # WEBHOOK_SECRET reaches curl through a 0600 config file, never argv โ€” see + # curlAuthPreamble, which always traps the removal. + CURL_CFG=$(mktemp) && chmod 600 "$CURL_CFG" + trap 'rm -f "$CURL_CFG"' EXIT + printf 'header = "Authorization: Bearer %s"\n' "$WEBHOOK_SECRET" > "$CURL_CFG" + # The scripts surface. load_script_manifest pins SCRIPTS_BASE_URL to it. + CI_SCRIPTS_URL="${HUB_BASE_URL%/}/api/ci/scripts" + + # _try_manifest โ€” 0 loaded, 1 no manifest surface there, 2 fatal. + # The manifest is asked for ONE group: its keys are the files to download. + _try_manifest() { + local base="$1" code + code=$(curl -sS -w '%{http_code}' -o "$SCRIPT_MANIFEST" \ + -K "$CURL_CFG" \ + "$base/manifest.json?group=$SCRIPT_GROUP") || code="000" + + if [ "$code" = "404" ]; then rm -f "$SCRIPT_MANIFEST"; return 1; fi + if [ "$code" != "200" ]; then + echo "โŒ manifest request to $base failed (HTTP $code)" + rm -f "$SCRIPT_MANIFEST" + return 2 + fi + # The digests are the TOP-LEVEL object. successResponse is the standard + # emitter but it does NOT add a wrapper โ€” it is NextResponse.json(data) + # plus the no-store header โ€” so there is no .data to reach through. + # A 200 that is not a manifest is how a hub which does not serve this path + # answers (the proxy rewrites unknown routes and returns HTML), so it + # means "wrong surface", not "corrupt". + if ! jq -e 'type == "object" and length > 0 and (to_entries | all(.value | type == "string"))' "$SCRIPT_MANIFEST" >/dev/null 2>&1; then + rm -f "$SCRIPT_MANIFEST" + return 1 + fi + return 0 + } + + # load_script_manifest + load_script_manifest() { + SCRIPT_GROUP="$1" + # "cmd; rc=$?" dies under the set -euo pipefail these steps run with โ€” + # errexit fires before rc is read and the step ends with NO output. And + # "if ! cmd; then rc=$?" is worse: inside the branch $? is the status of + # the NEGATION (0), so every failure reads as success. "|| rc=$?" is the + # one form that both suppresses errexit and preserves the real code. + local rc=0 + _try_manifest "$CI_SCRIPTS_URL" || rc=$? + if [ "$rc" = "0" ]; then + SCRIPTS_BASE_URL="$CI_SCRIPTS_URL" + echo "โœ… script manifest loaded ($(jq -r 'length' "$SCRIPT_MANIFEST") scripts)" + return 0 + fi + if [ "$rc" = "2" ]; then exit 1; fi + + # No manifest on the scripts surface: a hub older than the manifest + # itself. The manifest is the file list, so there is nothing to download. + echo "โŒ no script manifest on this hub ($HUB_BASE_URL): it cannot name the $SCRIPT_GROUP scripts. Redeploy the hub." + exit 1 + } + + # download_script_group โ€” the hub names the files, this workflow + # names only the group. Downloads every script of the group, in served order. + download_script_group() { + load_script_manifest "$1" + local name + # The loop runs in THIS shell (no pipe), so a failed download exits the step. + while IFS= read -r name; do + download_and_verify "$name" + done < <(jq -r 'keys_unsorted[]' "$SCRIPT_MANIFEST") + } + + download_and_verify() { + local script_name="$1" + local output_path="/tmp/$script_name" + + local expected_hash + expected_hash=$(jq -r --arg n "$script_name" '.[$n] // empty' "$SCRIPT_MANIFEST") + if [ -z "$expected_hash" ]; then + echo "โŒ $script_name is not in the server's script manifest!" + echo " The hub serves no such script, or it failed to read on the server." + exit 1 + fi + if ! printf '%s' "$expected_hash" | grep -Eq '^[0-9a-f]{64}$'; then + echo "โŒ the manifest entry for $script_name is not a SHA-256 digest โ€” refusing to run it." + exit 1 + fi + + curl -fsSL "$SCRIPTS_BASE_URL/$script_name" \ + -K "$CURL_CFG" \ + -o "$output_path" + + local actual_hash=$(shasum -a 256 "$output_path" | cut -d' ' -f1) + + if [ "$actual_hash" != "$expected_hash" ]; then + echo "โŒ HASH MISMATCH for $script_name!" + echo " Expected: $expected_hash" + echo " Actual: $actual_hash" + echo " The download was corrupted in transit โ€” both values come from the same deployment." + exit 1 + fi + + # Make shell scripts executable + if [[ "$script_name" == *.sh ]]; then + chmod +x "$output_path" + fi + + echo "โœ… $script_name verified (hash: ${actual_hash:0:16}...)" + } + + # Digests AND the file list come from the deployment serving the bytes, not from this file. + download_script_group "doc-helpers" + + # ========================================================================= + # REPORT RUN STARTED: the early "the workflow actually started" ping. + # Deliberately BEFORE the main script download: it stamps workflow_run_id + + # status 'running' on the hub's run row, which is what lets the hub's tiered + # reaper tell "dispatch accepted but nothing ran" (never pinged, failed + # fast) from "started and then crashed" (pinged, longer deadline). + # ========================================================================= + - name: Report run started + if: env.HUB_BASE_URL != '' + continue-on-error: true + env: + WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }} + run: | + source /tmp/workflow-helpers.sh + + CALLBACK_URL="${HUB_BASE_URL}/api/code-documentation/webhook" + echo "Reporting run start to $CALLBACK_URL" + + PAYLOAD="{ + \"run_id\": \"$RUN_ID\", + \"repo_id\": \"$REPO_ID\", + \"status\": \"running\", + \"workflow_run_id\": ${{ github.run_id }}, + \"workflow_url\": \"${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}\", + \"current_stage\": \"inline-docs\" + }" + + HTTP_CODE=$(send_webhook "$CALLBACK_URL" "$WEBHOOK_SECRET" "$PAYLOAD" "/tmp/webhook_start_response.txt") || HTTP_CODE="failed" + + if [ "$HTTP_CODE" = "200" ] || [ "$HTTP_CODE" = "201" ]; then + echo "Hub acknowledged the run start (HTTP $HTTP_CODE)" + else + echo "::warning title=Run start not reported::The hub answered HTTP $HTTP_CODE to the start callback. The run continues; the hub learns its status from the stage callbacks." + fi + + # ========================================================================= + # DOWNLOAD PIPELINE SCRIPTS + # Every script a documentation run uses, from the hub's authenticated + # /api/ci/scripts surface (workflow-helpers.sh arrived in "Download report + # helpers"), plus the source vocabulary and the Markdown guidelines. + # ========================================================================= + - name: Download pipeline scripts + id: helpers + env: + WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }} + # No HASH_* pins: the digests come from manifest.json on the same + # endpoint that serves the scripts, so a hash in this file can never + # be a different ref's than the bytes it checks. + run: | + echo "::group::Download the doc-pipeline script group" + + # Function to download and verify script + SCRIPT_MANIFEST=/tmp/flamingo-script-manifest.json + # WEBHOOK_SECRET reaches curl through a 0600 config file, never argv โ€” see + # curlAuthPreamble, which always traps the removal. + CURL_CFG=$(mktemp) && chmod 600 "$CURL_CFG" + trap 'rm -f "$CURL_CFG"' EXIT + printf 'header = "Authorization: Bearer %s"\n' "$WEBHOOK_SECRET" > "$CURL_CFG" + # The scripts surface. load_script_manifest pins SCRIPTS_BASE_URL to it. + CI_SCRIPTS_URL="${HUB_BASE_URL%/}/api/ci/scripts" + + # _try_manifest โ€” 0 loaded, 1 no manifest surface there, 2 fatal. + # The manifest is asked for ONE group: its keys are the files to download. + _try_manifest() { + local base="$1" code + code=$(curl -sS -w '%{http_code}' -o "$SCRIPT_MANIFEST" \ + -K "$CURL_CFG" \ + "$base/manifest.json?group=$SCRIPT_GROUP") || code="000" + + if [ "$code" = "404" ]; then rm -f "$SCRIPT_MANIFEST"; return 1; fi + if [ "$code" != "200" ]; then + echo "โŒ manifest request to $base failed (HTTP $code)" + rm -f "$SCRIPT_MANIFEST" + return 2 + fi + # The digests are the TOP-LEVEL object. successResponse is the standard + # emitter but it does NOT add a wrapper โ€” it is NextResponse.json(data) + # plus the no-store header โ€” so there is no .data to reach through. + # A 200 that is not a manifest is how a hub which does not serve this path + # answers (the proxy rewrites unknown routes and returns HTML), so it + # means "wrong surface", not "corrupt". + if ! jq -e 'type == "object" and length > 0 and (to_entries | all(.value | type == "string"))' "$SCRIPT_MANIFEST" >/dev/null 2>&1; then + rm -f "$SCRIPT_MANIFEST" + return 1 + fi + return 0 + } + + # load_script_manifest + load_script_manifest() { + SCRIPT_GROUP="$1" + # "cmd; rc=$?" dies under the set -euo pipefail these steps run with โ€” + # errexit fires before rc is read and the step ends with NO output. And + # "if ! cmd; then rc=$?" is worse: inside the branch $? is the status of + # the NEGATION (0), so every failure reads as success. "|| rc=$?" is the + # one form that both suppresses errexit and preserves the real code. + local rc=0 + _try_manifest "$CI_SCRIPTS_URL" || rc=$? + if [ "$rc" = "0" ]; then + SCRIPTS_BASE_URL="$CI_SCRIPTS_URL" + echo "โœ… script manifest loaded ($(jq -r 'length' "$SCRIPT_MANIFEST") scripts)" + return 0 + fi + if [ "$rc" = "2" ]; then exit 1; fi + + # No manifest on the scripts surface: a hub older than the manifest + # itself. The manifest is the file list, so there is nothing to download. + echo "โŒ no script manifest on this hub ($HUB_BASE_URL): it cannot name the $SCRIPT_GROUP scripts. Redeploy the hub." + exit 1 + } + + # download_script_group โ€” the hub names the files, this workflow + # names only the group. Downloads every script of the group, in served order. + download_script_group() { + load_script_manifest "$1" + local name + # The loop runs in THIS shell (no pipe), so a failed download exits the step. + while IFS= read -r name; do + download_and_verify "$name" + done < <(jq -r 'keys_unsorted[]' "$SCRIPT_MANIFEST") + } + + download_and_verify() { + local script_name="$1" + local output_path="/tmp/$script_name" + + local expected_hash + expected_hash=$(jq -r --arg n "$script_name" '.[$n] // empty' "$SCRIPT_MANIFEST") + if [ -z "$expected_hash" ]; then + echo "โŒ $script_name is not in the server's script manifest!" + echo " The hub serves no such script, or it failed to read on the server." + exit 1 + fi + if ! printf '%s' "$expected_hash" | grep -Eq '^[0-9a-f]{64}$'; then + echo "โŒ the manifest entry for $script_name is not a SHA-256 digest โ€” refusing to run it." + exit 1 + fi + + curl -fsSL "$SCRIPTS_BASE_URL/$script_name" \ + -K "$CURL_CFG" \ + -o "$output_path" + + local actual_hash=$(shasum -a 256 "$output_path" | cut -d' ' -f1) + + if [ "$actual_hash" != "$expected_hash" ]; then + echo "โŒ HASH MISMATCH for $script_name!" + echo " Expected: $expected_hash" + echo " Actual: $actual_hash" + echo " The download was corrupted in transit โ€” both values come from the same deployment." + exit 1 + fi + + # Make shell scripts executable + if [[ "$script_name" == *.sh ]]; then + chmod +x "$output_path" + fi + + echo "โœ… $script_name verified (hash: ${actual_hash:0:16}...)" + } + + # Every script a documentation run uses, stage 0 (the code graph) included. + # Digests AND the file list come from the deployment serving the bytes, + # not from this file: the hub lists the group in download order (a helper + # a generator require()s at load comes before it). + download_script_group "doc-pipeline" + echo "::endgroup::" + + # ONE vocabulary read for the whole run: every later step (language + # detection, source discovery, the generators, the graph build) reads this + # file, so none of them needs the secret for it. + node /tmp/ci-source.mjs vocabulary /tmp/ci-vocabulary.json || { echo "::error title=Source vocabulary unavailable::Could not read the source vocabulary from the hub."; exit 1; } + echo "CODE_GRAPH_VOCABULARY_FILE=/tmp/ci-vocabulary.json" >> $GITHUB_ENV + + echo "Pipeline scripts and source vocabulary downloaded and verified" + + # Export paths for all stages (use os.tmpdir() compatible paths) + echo "VALIDATION_RULES_PATH=/tmp/markdown-validation-rules.md" >> $GITHUB_ENV + echo "GUIDELINES_PATH=/tmp/flamingo-markdown-guidelines.md" >> $GITHUB_ENV + echo "STAGE3_FILES_TRACKER=/tmp/stage3-files.txt" >> $GITHUB_ENV + echo "STAGE3_STATS_FILE=/tmp/.doc-stage3-stats.json" >> $GITHUB_ENV + echo "STAGE4_FILES_TRACKER=/tmp/stage4-files.txt" >> $GITHUB_ENV + + # Flamingo Markdown guidelines (their own endpoint). REQUIRED: the + # Markdown validation and the CodeWiki prompts both read them. + GUIDELINES_URL="${HUB_BASE_URL}/api/code-documentation/guidelines" + # Same 0600 config file the download block above set up. + HTTP_CODE=$(curl -fsSL -w "%{http_code}" \ + "$GUIDELINES_URL" \ + -K "$CURL_CFG" \ + -o "/tmp/flamingo-markdown-guidelines.md" 2>/dev/null) || HTTP_CODE="failed" + + if [ "$HTTP_CODE" = "200" ]; then + GUIDELINES_SIZE=$(wc -c < /tmp/flamingo-markdown-guidelines.md | tr -d ' ') + if [ "$GUIDELINES_SIZE" -lt 100 ]; then + echo "::error title=Markdown guidelines invalid::The guidelines file is $GUIDELINES_SIZE bytes, which is an error response, not guidelines. Its body follows." + cat /tmp/flamingo-markdown-guidelines.md + exit 1 + fi + echo "Markdown guidelines downloaded ($GUIDELINES_SIZE bytes)" + else + echo "::error title=Markdown guidelines unavailable::GET $GUIDELINES_URL answered HTTP $HTTP_CODE. The Markdown validation and the CodeWiki prompts require them; check that the hub serves the guidelines endpoint." + rm -f /tmp/flamingo-markdown-guidelines.md + exit 1 + fi + + # (The run-started report sits ABOVE the main script download; see the + # report-capability step ordering at the top of the job.) + + - name: Check out repository + # v5 = the Node 24 drop-in (v4 targets EOL Node 20 and warns on every run). + uses: actions/checkout@fbc6f3992d24b796d5a048ff273f7fcc4a7b6c09 # v5.1.0 + with: + fetch-depth: 0 + token: ${{ secrets.GITHUB_TOKEN }} + ref: ${{ env.SOURCE_BRANCH }} + + # The SOURCE head, before the PR branch and the docs-removal commit move + # HEAD: the stage-0 graph build tags this commit (the code being + # documented), never the docs branch it is sitting on. + - name: Record source head + run: echo "SOURCE_HEAD_SHA=$(git rev-parse HEAD)" >> "$GITHUB_ENV" + + # ========================================================================= + # DEPENDENCY REPOSITORIES (when configured): cloned beside the checkout as + # context for every stage. + # ========================================================================= + - name: Clone dependency repositories + if: env.DEPENDENCIES != '' + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + # Authenticates the clone-token request to the hub (masked by the runner). + WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }} + run: | + source /tmp/workflow-helpers.sh + + echo "Cloning dependency repositories: $DEPENDENCIES" + ensure_directory "../deps" + + # The clone credential is minted by the hub for THIS run: read-only, + # scoped to exactly this repository's dependencies, valid one hour. + # A stored repository secret cannot hold it (an installation token + # expires an hour after it is written). + CLONE_TOKEN="" + TOKEN_FILE=$(mktemp) && chmod 600 "$TOKEN_FILE" + HTTP_CODE=$(node /tmp/ci-hub.mjs get "/api/ci/code-documentation/clone-token?github=${GITHUB_REPOSITORY}" "$TOKEN_FILE") || HTTP_CODE="000" + if [ "$HTTP_CODE" = "200" ]; then + CLONE_TOKEN=$(jq -r '.token // empty' "$TOKEN_FILE") + fi + rm -f "$TOKEN_FILE" + if [ -n "$CLONE_TOKEN" ]; then + echo "::add-mask::$CLONE_TOKEN" + echo "Credential: a read-only token the hub minted for this run's dependencies" + else + echo "::warning title=No clone token::The hub did not mint a clone token (HTTP $HTTP_CODE); using GITHUB_TOKEN, which reads public repositories only." + CLONE_TOKEN="$GH_TOKEN" + fi + + # The credential applies to THESE clones only (`git -c`), never globally: + # the token reads the dependencies and nothing else, so a global URL + # rewrite would also send this job's own pushes through it. + IFS=',' read -ra DEPS <<< "$DEPENDENCIES" + for dep in "${DEPS[@]}"; do + repo_name=$(basename $dep) + echo "::group::Clone $dep into ../deps/$repo_name" + if git -c "url.https://x-access-token:${CLONE_TOKEN}@github.com/.insteadOf=https://github.com/" \ + clone --depth 1 "https://github.com/$dep.git" "../deps/$repo_name" 2>&1; then + git -C "../deps/$repo_name" remote set-url origin "https://github.com/$dep.git" + file_count=$(node /tmp/ci-source.mjs count "../deps/$repo_name" 2>/dev/null || echo "?") + echo "Cloned $dep ($file_count source files)" + else + echo "::warning title=Dependency not cloned::Could not clone $dep." + fi + echo "::endgroup::" + done + + echo "::group::Dependency directories" + ls -la ../deps/ 2>/dev/null || echo "No dependencies cloned" + echo "::endgroup::" + echo "Dependency source files available as context: $(node /tmp/ci-source.mjs count ../deps 2>/dev/null || echo '?')" + + # ========================================================================= + # PRIMARY LANGUAGE (before every stage, so all of them agree): picks the + # stage 2 engine and filters stages 1 to 3. + # ========================================================================= + - name: Detect primary language + id: detect_language + run: | + source /tmp/workflow-helpers.sh + + # This step is the FIRST reader of CODEWIKI_CONFIG_JSON โ€” it runs before + # "Validate and parse run configuration", so the emptiness guard lives here, + # ahead of the first jq, rather than in the later validation step. + if [ -z "$CODEWIKI_CONFIG_JSON" ]; then + echo "::error title=Missing configuration::CODEWIKI_CONFIG_JSON is empty; language detection needs it." + exit 1 + fi + + echo "Detecting the primary language across the repository (.) and its dependencies (../deps/)" + + # ONE detection, from the vocabulary the hub serves (ci-source.mjs): which + # languages are source, their extensions, and which of them CodeWiki can + # parse are rows in the hub's language table. This step used to carry nine + # hand-typed `find` counts, a positional helper and a six-way threshold + # test, each with its own idea of the extensions and the exclusions. + DETECTION=$(node /tmp/ci-source.mjs detect) || { echo "::error title=Language detection failed::ci-source.mjs detect exited non-zero."; exit 1; } + PRIMARY_LANG=$(echo "$DETECTION" | jq -r '.primary') + MAX_COUNT=$(echo "$DETECTION" | jq -r '.max') + CODEWIKI_SUPPORTED=$(echo "$DETECTION" | jq -r '.codewiki_supported') + + echo "::group::Source files by language (tests and never-source directories excluded)" + echo "$DETECTION" | jq -r '.counts | to_entries[] | select(.value > 0) | " \(.key): \(.value)"' + echo "::endgroup::" + echo "Primary language: $PRIMARY_LANG ($MAX_COUNT files)" + if [ "$CODEWIKI_SUPPORTED" = "true" ]; then + echo "CodeWiki can parse it: yes" + else + echo "CodeWiki can parse it: no (stage 2 uses the Claude architecture analysis)" + fi + + # Per-repo engine override. + # + # The detection above cannot see mixed repos: it counts `.` AND `../deps`, so + # a Rust or Go product with a TypeScript dependency clones its way past the + # >=10 threshold and runs CodeWiki over a codebase whose analyzers do not + # exist โ€” which yields synthetic module_1/module_2/... docs that look like a + # successful run. `engine` pins the choice. + # + # Applied here, before set_output, so all five downstream gates keep reading + # one value and need no change. It cannot live in the `if:` conditions: + # GitHub Actions expressions have no ternary. + CODEWIKI_ENGINE=$(require_json_key "$CODEWIKI_CONFIG_JSON" '.engine' 'codewiki engine') || exit 1 + case "$CODEWIKI_ENGINE" in + claude) + CODEWIKI_SUPPORTED="false" + echo "Stage 2 engine: claude (pinned by the repository configuration)" + ;; + codewiki) + CODEWIKI_SUPPORTED="true" + echo "Stage 2 engine: codewiki (pinned by the repository configuration)" + ;; + auto) + echo "Stage 2 engine: auto (CodeWiki supported: $CODEWIKI_SUPPORTED)" + ;; + *) + echo "::error title=Invalid stage 2 engine::engine is '$CODEWIKI_ENGINE'; expected auto, codewiki or claude." + exit 1 + ;; + esac + + # Output for use by subsequent steps + set_output "primary_language" "$PRIMARY_LANG" + set_output "codewiki_supported" "$CODEWIKI_SUPPORTED" + set_output "file_count" "$MAX_COUNT" + + # ========================================================================= + # VALIDATE AND PARSE THE RUN CONFIGURATION + # 1. Every required parameter is present + # 2. Each JSON configuration is parsed into individual variables (the JSON + # grouping keeps the dispatch under GitHub's 25-input limit) + # 3. The parsed values are valid, then exported to GITHUB_ENV + # ========================================================================= + - name: Validate and parse run configuration + run: | + # 1. Required parameters ------------------------------------------- + VALIDATION_FAILED=0 + + # Core parameters + [ -z "$RUN_ID" ] && echo "::error title=Missing parameter::RUN_ID" && VALIDATION_FAILED=1 + [ -z "$REPO_ID" ] && echo "::error title=Missing parameter::REPO_ID" && VALIDATION_FAILED=1 + [ -z "$HUB_BASE_URL" ] && echo "::error title=Missing parameter::HUB_BASE_URL" && VALIDATION_FAILED=1 + + [ -z "$STAGES" ] && echo "::error title=Missing parameter::STAGES" && VALIDATION_FAILED=1 + [ -z "$CLAUDE_MODEL" ] && echo "::error title=Missing parameter::CLAUDE_MODEL" && VALIDATION_FAILED=1 + + # JSON parameters + [ -z "$CODEWIKI_CONFIG_JSON" ] && echo "::error title=Missing parameter::CODEWIKI_CONFIG_JSON" && VALIDATION_FAILED=1 + [ -z "$OUTPUT_PATHS_JSON" ] && echo "::error title=Missing parameter::OUTPUT_PATHS_JSON" && VALIDATION_FAILED=1 + [ -z "$README_CONFIG_JSON" ] && echo "::error title=Missing parameter::README_CONFIG_JSON" && VALIDATION_FAILED=1 + [ -z "$YOUTUBE_CONFIG_JSON" ] && echo "::error title=Missing parameter::YOUTUBE_CONFIG_JSON" && VALIDATION_FAILED=1 + + if [ $VALIDATION_FAILED -eq 1 ]; then + echo "::error title=Invalid run configuration::Required parameters are missing (annotated above). The hub sends every one of them; re-run \"Setup Workflow\" if this repository's workflow is out of date." + exit 1 + fi + + echo "Required parameters: present" + + # 2a. CodeWiki configuration (nested JSON) --------------------------- + echo "::group::CodeWiki configuration" + + # Only the keys with a real consumer are extracted here โ€” the per-phase + # base_url / api_version / temperature / temperature_supported are read + # directly from CODEWIKI_CONFIG_JSON by configure_codewiki_from_json, which + # is the single place that builds the `codewiki config set` command. They + # used to be parsed here as well and exported to $GITHUB_ENV, where nothing + # read them. + # + # No `// default` fallbacks anywhere below. The hub deep-merges every JSON + # param against the params SSOT before dispatch, so an absent key is a real + # bug โ€” and a fallback here would silently win over the SSOT, which is how + # docs/architecture and docs/reference/architecture drifted apart. + # `require_json_key` / `optional_json_key` come from workflow-helpers.sh. + source /tmp/workflow-helpers.sh + CW_JSON="$CODEWIKI_CONFIG_JSON" + + # Parse nested cluster config + CODEWIKI_CLUSTER_PROVIDER=$(require_json_key "$CW_JSON" '.cluster.provider' 'cluster provider') || exit 1 + CODEWIKI_CLUSTER_MODEL=$(require_json_key "$CW_JSON" '.cluster.model' 'cluster model') || exit 1 + CODEWIKI_CLUSTER_MAX_TOKENS=$(require_json_key "$CW_JSON" '.cluster.max_tokens' 'cluster max_tokens') || exit 1 + CODEWIKI_CLUSTER_MAX_TOKEN_FIELD=$(require_json_key "$CW_JSON" '.cluster.max_token_field' 'cluster max_token_field') || exit 1 + # Nullable by design: api_version is null for every OpenAI model. + + # Parse nested generation config + CODEWIKI_GENERATION_PROVIDER=$(require_json_key "$CW_JSON" '.generation.provider' 'generation provider') || exit 1 + CODEWIKI_GENERATION_MODEL=$(require_json_key "$CW_JSON" '.generation.model' 'generation model') || exit 1 + CODEWIKI_GENERATION_MAX_TOKENS=$(require_json_key "$CW_JSON" '.generation.max_tokens' 'generation max_tokens') || exit 1 + CODEWIKI_GENERATION_MAX_TOKEN_FIELD=$(require_json_key "$CW_JSON" '.generation.max_token_field' 'generation max_token_field') || exit 1 + + # Parse nested fallback config + CODEWIKI_FALLBACK_PROVIDER=$(require_json_key "$CW_JSON" '.fallback.provider' 'fallback provider') || exit 1 + CODEWIKI_FALLBACK_MODEL=$(require_json_key "$CW_JSON" '.fallback.model' 'fallback model') || exit 1 + CODEWIKI_FALLBACK_MAX_TOKENS=$(require_json_key "$CW_JSON" '.fallback.max_tokens' 'fallback max_tokens') || exit 1 + CODEWIKI_FALLBACK_MAX_TOKEN_FIELD=$(require_json_key "$CW_JSON" '.fallback.max_token_field' 'fallback max_token_field') || exit 1 + + # Parse top-level config. + # max_files_per_module is LIVE: this value reaches CodeWiki through the + # job-scoped $GITHUB_ENV write below, and upstream reads it in its + # empty-module-tree branch โ€” the branch Go/Rust/HCL repos land in. + CODEWIKI_MAX_FILES_PER_MODULE=$(require_json_key "$CW_JSON" '.max_files_per_module' 'max_files_per_module') || exit 1 + CODEWIKI_MAX_DEPTH=$(require_json_key "$CW_JSON" '.max_depth' 'max_depth') || exit 1 + CODEWIKI_REPO=$(require_json_key "$CW_JSON" '.repo' 'codewiki repo url') || exit 1 + + echo "Cluster (phase 2): $CODEWIKI_CLUSTER_PROVIDER / $CODEWIKI_CLUSTER_MODEL (${CODEWIKI_CLUSTER_MAX_TOKEN_FIELD}, ${CODEWIKI_CLUSTER_MAX_TOKENS} tokens)" + echo "Generation (phase 3+): $CODEWIKI_GENERATION_PROVIDER / $CODEWIKI_GENERATION_MODEL (${CODEWIKI_GENERATION_MAX_TOKEN_FIELD}, ${CODEWIKI_GENERATION_MAX_TOKENS} tokens)" + echo "Fallback: $CODEWIKI_FALLBACK_PROVIDER / $CODEWIKI_FALLBACK_MODEL (${CODEWIKI_FALLBACK_MAX_TOKEN_FIELD}, ${CODEWIKI_FALLBACK_MAX_TOKENS} tokens)" + echo "Max depth: $CODEWIKI_MAX_DEPTH" + echo "Max files per module: $CODEWIKI_MAX_FILES_PER_MODULE" + echo "::endgroup::" + + # 2b. YouTube configuration ------------------------------------------ + echo "::group::YouTube configuration" + + # Parse YouTube config from JSONB (channels only - API key from secrets) + # channels is legitimately optional: no channels == feature off + YOUTUBE_CHANNELS=$(echo "$YOUTUBE_CONFIG_JSON" | jq -c '.channels // []') + + # YouTube is enabled if channels array has items + YOUTUBE_ENABLED=$(echo "$YOUTUBE_CHANNELS" | jq -r 'if length > 0 then "true" else "false" end') + + echo "Enabled: $YOUTUBE_ENABLED (from the channel count)" + echo "Channels: $YOUTUBE_CHANNELS" + echo "API key: secrets.YOUTUBE_API_KEY (never stored on the hub)" + echo "::endgroup::" + + # 2c. README logo configuration -------------------------------------- + echo "::group::README logo configuration" + + # Parse README config from JSONB + README_LOGO_DARK=$(optional_json_key "$README_CONFIG_JSON" '.logo_dark') + README_LOGO_LIGHT=$(optional_json_key "$README_CONFIG_JSON" '.logo_light') + README_LOGO_ALT=$(require_json_key "$README_CONFIG_JSON" '.logo_alt' 'readme logo alt') || exit 1 + + echo "Dark logo: ${README_LOGO_DARK:-'(not set)'}" + echo "Light logo: ${README_LOGO_LIGHT:-'(not set)'}" + echo "Alt text: $README_LOGO_ALT" + echo "::endgroup::" + + # 2d. Output paths --------------------------------------------------- + echo "::group::Output paths" + + # Same fail-loud contract as the codewiki block. These fallbacks were the + # last surviving copy of the five paths, and `.reference` still said + # docs/architecture while the SSOT said docs/reference/architecture โ€” the + # very drift the SSOT was created to end. + OP_JSON="$OUTPUT_PATHS_JSON" + DOCS_OUTPUT_PATH=$(require_json_key "$OP_JSON" '.docs' 'docs output path') || exit 1 + REFERENCE_OUTPUT_PATH=$(require_json_key "$OP_JSON" '.reference' 'reference output path') || exit 1 + DIAGRAMS_OUTPUT_PATH=$(require_json_key "$OP_JSON" '.diagrams' 'diagrams output path') || exit 1 + GETTING_STARTED_OUTPUT_PATH=$(require_json_key "$OP_JSON" '.getting_started' 'getting-started output path') || exit 1 + DEVELOPMENT_OUTPUT_PATH=$(require_json_key "$OP_JSON" '.development' 'development output path') || exit 1 + + echo "Docs root: $DOCS_OUTPUT_PATH" + echo "Reference: $REFERENCE_OUTPUT_PATH" + echo "Diagrams: $DIAGRAMS_OUTPUT_PATH" + echo "Getting started: $GETTING_STARTED_OUTPUT_PATH" + echo "Development: $DEVELOPMENT_OUTPUT_PATH" + echo "::endgroup::" + + # 2e. Custom instructions and external repositories ------------------ + echo "::group::Custom instructions and external repositories" + + # Repository context (prevents invented GitHub URLs in every prompt) + # Extract repository information from GitHub context + GITHUB_REPO="${{ github.repository }}" + GITHUB_OWNER="${{ github.repository_owner }}" + GITHUB_SERVER="${{ github.server_url }}" + GITHUB_REPO_NAME=$(echo "$GITHUB_REPO" | cut -d'/' -f2) + GITHUB_REPO_URL="${GITHUB_SERVER}/${GITHUB_REPO}" + + # Build repository context section (injected into ALL AI prompts) + # Use printf for multi-line string (avoids YAML parsing issues with heredoc) + printf -v REPOSITORY_CONTEXT '%s\n' \ + '## REPOSITORY CONTEXT - GROUND TRUTH' \ + '' \ + '**CRITICAL:** This section provides the ACTUAL repository information. You MUST use these exact values when constructing GitHub URLs.' \ + '' \ + "- **Repository:** ${GITHUB_REPO}" \ + "- **Owner:** ${GITHUB_OWNER}" \ + "- **Repository Name:** ${GITHUB_REPO_NAME}" \ + "- **Repository URL:** ${GITHUB_REPO_URL}" \ + "- **Server:** ${GITHUB_SERVER}" \ + '' \ + '**MANDATORY RULES FOR GITHUB URLS:**' \ + "1. ALWAYS use the exact repository path: \`${GITHUB_REPO}\`" \ + '2. NEVER use placeholder URLs like "your-org", "example-org", or "mycompany"' \ + '3. NEVER infer repository owner from file contents or dependencies' \ + '4. NEVER use upstream/parent repository URLs (if this is a fork, use the fork URL)' \ + "5. When linking to code: \`${GITHUB_REPO_URL}/blob/main/path/to/file\`" \ + "6. When linking to clone: \`git clone ${GITHUB_REPO_URL}.git\`" \ + "7. When linking to issues/PRs: \`${GITHUB_REPO_URL}/issues\` or \`${GITHUB_REPO_URL}/pulls\`" \ + "8. When linking to releases: \`${GITHUB_REPO_URL}/releases\`" \ + '' \ + '**If you find yourself writing a GitHub URL, verify it matches the Repository URL above.**' \ + "**ESPECIALLY IN README.md and tutorials - All GitHub URLs MUST use ${GITHUB_REPO}**" \ + '' \ + '---' + + echo "Repository: $GITHUB_REPO" + echo "Repository URL: $GITHUB_REPO_URL" + + # Repository context goes in front of the custom instructions + # Custom instructions come as plain text from user + # Read from the environment rather than interpolating into single quotes. + # GitHub Actions expression substitution runs BEFORE bash parses the line, so a + # single apostrophe anywhere in an admin's instructions used to terminate the + # string and kill the step with a syntax error. NOTE: never write a literal + # empty GitHub expression in this run block, even inside a comment โ€” Actions + # evaluates it pre-bash and the whole workflow fails to parse. + USER_CUSTOM_INSTRUCTIONS="$CUSTOM_REPO_INSTRUCTIONS" + + # Combine repository context + user custom instructions + # Repository context goes FIRST (highest priority in prompts) + if [ -n "$USER_CUSTOM_INSTRUCTIONS" ]; then + CUSTOM_INSTRUCTIONS="${REPOSITORY_CONTEXT}"$'\n\n'"${USER_CUSTOM_INSTRUCTIONS}" + else + CUSTOM_INSTRUCTIONS="$REPOSITORY_CONTEXT" + fi + + # External repos come as separate JSON array parameter + + EXTERNAL_REPOS_COUNT=$(echo "$EXTERNAL_REPOS" | jq '. | length' 2>/dev/null || echo "0") + echo "Repository context: ${#REPOSITORY_CONTEXT} chars" + echo "Custom instructions: ${#USER_CUSTOM_INSTRUCTIONS} chars" + echo "Combined prompt block: ${#CUSTOM_INSTRUCTIONS} chars" + echo "External repositories: $EXTERNAL_REPOS_COUNT" + echo "Combined prompt block, first 300 chars:" + echo "${CUSTOM_INSTRUCTIONS:0:300}..." + echo "::endgroup::" + + # 3. Parsed values --------------------------------------------------- + + # Validate providers + if [[ ! "$CODEWIKI_CLUSTER_PROVIDER" =~ ^(anthropic|openai)$ ]]; then + echo "::error title=Invalid CodeWiki configuration::cluster provider is '$CODEWIKI_CLUSTER_PROVIDER'; expected anthropic or openai." + exit 1 + fi + + if [[ ! "$CODEWIKI_GENERATION_PROVIDER" =~ ^(anthropic|openai)$ ]]; then + echo "::error title=Invalid CodeWiki configuration::generation provider is '$CODEWIKI_GENERATION_PROVIDER'; expected anthropic or openai." + exit 1 + fi + + if [[ ! "$CODEWIKI_FALLBACK_PROVIDER" =~ ^(anthropic|openai)$ ]]; then + echo "::error title=Invalid CodeWiki configuration::fallback provider is '$CODEWIKI_FALLBACK_PROVIDER'; expected anthropic or openai." + exit 1 + fi + + # Model names are not empty + [ -z "$CODEWIKI_CLUSTER_MODEL" ] && echo "::error title=Invalid CodeWiki configuration::cluster model is empty." && exit 1 + [ -z "$CODEWIKI_GENERATION_MODEL" ] && echo "::error title=Invalid CodeWiki configuration::generation model is empty." && exit 1 + [ -z "$CODEWIKI_FALLBACK_MODEL" ] && echo "::error title=Invalid CodeWiki configuration::fallback model is empty." && exit 1 + + # Numeric values + [[ ! "$CODEWIKI_CLUSTER_MAX_TOKENS" =~ ^[0-9]+$ ]] && echo "::error title=Invalid CodeWiki configuration::cluster max_tokens is '$CODEWIKI_CLUSTER_MAX_TOKENS', not a number." && exit 1 + [[ ! "$CODEWIKI_GENERATION_MAX_TOKENS" =~ ^[0-9]+$ ]] && echo "::error title=Invalid CodeWiki configuration::generation max_tokens is '$CODEWIKI_GENERATION_MAX_TOKENS', not a number." && exit 1 + [[ ! "$CODEWIKI_FALLBACK_MAX_TOKENS" =~ ^[0-9]+$ ]] && echo "::error title=Invalid CodeWiki configuration::fallback max_tokens is '$CODEWIKI_FALLBACK_MAX_TOKENS', not a number." && exit 1 + [[ ! "$CODEWIKI_MAX_DEPTH" =~ ^[0-9]+$ ]] && echo "::error title=Invalid CodeWiki configuration::max_depth is '$CODEWIKI_MAX_DEPTH', not a number." && exit 1 + [[ ! "$CODEWIKI_MAX_FILES_PER_MODULE" =~ ^[0-9]+$ ]] && echo "::error title=Invalid CodeWiki configuration::max_files_per_module is '$CODEWIKI_MAX_FILES_PER_MODULE', not a number." && exit 1 + + echo "Parsed values: valid" + + # 4. Export to GITHUB_ENV (every later step reads these) ------------- + + # CodeWiki config (31 vars: cluster=9, generation=10, fallback=9, shared=3) + echo "CODEWIKI_CLUSTER_PROVIDER=$CODEWIKI_CLUSTER_PROVIDER" >> $GITHUB_ENV + echo "CODEWIKI_CLUSTER_MODEL=$CODEWIKI_CLUSTER_MODEL" >> $GITHUB_ENV + echo "CODEWIKI_CLUSTER_MAX_TOKENS=$CODEWIKI_CLUSTER_MAX_TOKENS" >> $GITHUB_ENV + echo "CODEWIKI_CLUSTER_MAX_TOKEN_FIELD=$CODEWIKI_CLUSTER_MAX_TOKEN_FIELD" >> $GITHUB_ENV + echo "CODEWIKI_GENERATION_PROVIDER=$CODEWIKI_GENERATION_PROVIDER" >> $GITHUB_ENV + echo "CODEWIKI_GENERATION_MODEL=$CODEWIKI_GENERATION_MODEL" >> $GITHUB_ENV + echo "CODEWIKI_GENERATION_MAX_TOKENS=$CODEWIKI_GENERATION_MAX_TOKENS" >> $GITHUB_ENV + echo "CODEWIKI_GENERATION_MAX_TOKEN_FIELD=$CODEWIKI_GENERATION_MAX_TOKEN_FIELD" >> $GITHUB_ENV + echo "CODEWIKI_FALLBACK_PROVIDER=$CODEWIKI_FALLBACK_PROVIDER" >> $GITHUB_ENV + echo "CODEWIKI_FALLBACK_MODEL=$CODEWIKI_FALLBACK_MODEL" >> $GITHUB_ENV + echo "CODEWIKI_FALLBACK_MAX_TOKENS=$CODEWIKI_FALLBACK_MAX_TOKENS" >> $GITHUB_ENV + echo "CODEWIKI_FALLBACK_MAX_TOKEN_FIELD=$CODEWIKI_FALLBACK_MAX_TOKEN_FIELD" >> $GITHUB_ENV + echo "CODEWIKI_MAX_FILES_PER_MODULE=$CODEWIKI_MAX_FILES_PER_MODULE" >> $GITHUB_ENV + echo "CODEWIKI_MAX_DEPTH=$CODEWIKI_MAX_DEPTH" >> $GITHUB_ENV + echo "CODEWIKI_REPO=$CODEWIKI_REPO" >> $GITHUB_ENV + + # YouTube config (2 vars - API key from secrets, not exported here) + echo "YOUTUBE_ENABLED=$YOUTUBE_ENABLED" >> $GITHUB_ENV + echo "YOUTUBE_CHANNELS=$YOUTUBE_CHANNELS" >> $GITHUB_ENV + + # README config (3 vars) + echo "README_LOGO_DARK=$README_LOGO_DARK" >> $GITHUB_ENV + echo "README_LOGO_LIGHT=$README_LOGO_LIGHT" >> $GITHUB_ENV + echo "README_LOGO_ALT=$README_LOGO_ALT" >> $GITHUB_ENV + + # Output paths (5 vars) + echo "DOCS_OUTPUT_PATH=$DOCS_OUTPUT_PATH" >> $GITHUB_ENV + echo "REFERENCE_OUTPUT_PATH=$REFERENCE_OUTPUT_PATH" >> $GITHUB_ENV + echo "DIAGRAMS_OUTPUT_PATH=$DIAGRAMS_OUTPUT_PATH" >> $GITHUB_ENV + echo "GETTING_STARTED_OUTPUT_PATH=$GETTING_STARTED_OUTPUT_PATH" >> $GITHUB_ENV + echo "DEVELOPMENT_OUTPUT_PATH=$DEVELOPMENT_OUTPUT_PATH" >> $GITHUB_ENV + + # Custom instructions (2 vars) + echo "CUSTOM_INSTRUCTIONS<> $GITHUB_ENV + echo "$CUSTOM_INSTRUCTIONS" >> $GITHUB_ENV + echo "EOF" >> $GITHUB_ENV + echo "EXTERNAL_REPOS=$EXTERNAL_REPOS" >> $GITHUB_ENV + + echo "Run configuration validated and exported: 15 CodeWiki, 2 YouTube, 3 README, 5 output paths, 2 instruction values" + + # ========================================================================= + # SOURCE FILE DISCOVERY (the one list every stage reads) + # Discovers SOURCE CODE files and optionally DELETES everything else. + # When SOURCE_FILES_LIMIT > 0: + # 1. Keeps only N source files (.ts, .java, .py, etc.) + # 2. DELETES ALL other files in the repo (aggressive cleanup) + # Generated docs (inline .md) are created AFTER this step, so not affected. + # ========================================================================= + - name: Discover source files + id: discover_files + env: + SOURCE_FILES_LIMIT: ${{ env.SOURCE_FILES_LIMIT }} + # DOCS_OUTPUT_PATH is needed below so the find can exclude the + # generated docs tree (deleted by the docs-removal step that runs + # AFTER discovery but BEFORE Stage 1 โ€” source-extension files + # under docs/ would otherwise be enumerated, then deleted, then + # cause ENOENT in Stage 1). + DOCS_OUTPUT_PATH: ${{ env.DOCS_OUTPUT_PATH }} + run: | + source /tmp/workflow-helpers.sh + + echo "Discovering source files in the repository (.) and its dependencies (../deps/)" + + # File paths (all in /tmp to avoid accidental commits) + SOURCE_FILES_LIST="/tmp/.code-documentation-source-files.txt" + ALL_SOURCE_TEMP="/tmp/all_source_files_discovered.txt" + ALL_FILES_TEMP="/tmp/all_files_in_repo.txt" + FILES_TO_DELETE="/tmp/files_to_delete.txt" + + # 1. Find SOURCE CODE files only (with standard exclusions) + # IMPORTANT: exclude $DOCS_OUTPUT_PATH/* โ€” the docs-removal step below + # does `rm -rf $DOCS_OUTPUT_PATH` BEFORE Stage 1 reads this list. If + # any source-extension file lives under the docs tree (Doxygen's + # `docs/doxygen/documentation.h`, Sphinx `_extensions/*.py`, etc.), + # it would be enumerated here, then deleted, then Stage 1 hits + # ENOENT trying to read it. This is the source-discovery / + # docs-removal / Stage-1 ordering bug โ€” exclusion is the targeted fix. + # The list comes from ci-source.mjs: the served source extensions, the + # served never-source directories, tests skipped by name โ€” the same rule + # the language detection above counted with. + node /tmp/ci-source.mjs list "$ALL_SOURCE_TEMP" --exclude "./$DOCS_OUTPUT_PATH" > /dev/null || { echo "::error title=Source discovery failed::ci-source.mjs list exited non-zero."; exit 1; } + + TOTAL_SOURCE=$(wc -l < "$ALL_SOURCE_TEMP" | tr -d ' ') + FILE_LIMIT="${SOURCE_FILES_LIMIT:-0}" + + echo "Source files found: $TOTAL_SOURCE" + + # Apply limit: keep N source files, DELETE EVERYTHING ELSE + if [ "$FILE_LIMIT" -gt 0 ]; then + echo "::warning title=Source file limit active::SOURCE_FILES_LIMIT=$FILE_LIMIT. Keeping $FILE_LIMIT source files and deleting every other file in the checkout (a debugging setting)." + echo "::group::Apply the source file limit" + + # Keep first N source files + head -n "$FILE_LIMIT" "$ALL_SOURCE_TEMP" > "$SOURCE_FILES_LIST" + KEEPING=$(wc -l < "$SOURCE_FILES_LIST" | tr -d ' ') + + # 2. Find ALL files in the repo (except .git and workflow temp files) + find . ../deps 2>/dev/null -type f \ + -not -path "*/.git/*" \ + -not -path "*/.git" \ + -not -name ".code-documentation-*" \ + -not -name ".doc-stage*" \ + | sort > "$ALL_FILES_TEMP" + + TOTAL_FILES=$(wc -l < "$ALL_FILES_TEMP" | tr -d ' ') + echo "Files in the checkout: $TOTAL_FILES" + + # Build delete list: ALL files EXCEPT the ones we're keeping + # Also preserve workflow temp files (.code-documentation-*, .doc-stage*) + > "$FILES_TO_DELETE" + while IFS= read -r file; do + # Skip workflow temp files we need to preserve + case "$file" in + ./.code-documentation-*|./.doc-stage*) continue ;; + esac + # Check if this file is in our keep list + if ! grep -qxF "$file" "$SOURCE_FILES_LIST" 2>/dev/null; then + echo "$file" >> "$FILES_TO_DELETE" + fi + done < "$ALL_FILES_TEMP" + + DELETE_COUNT=$(wc -l < "$FILES_TO_DELETE" | tr -d ' ') + echo "Files to delete: $DELETE_COUNT" + + # Delete all files NOT in the keep list + DELETED_COUNT=0 + while IFS= read -r file_to_delete; do + if [ -f "$file_to_delete" ]; then + rm -f "$file_to_delete" + DELETED_COUNT=$((DELETED_COUNT + 1)) + fi + done < "$FILES_TO_DELETE" + + echo "Kept $KEEPING source files, deleted $DELETED_COUNT files" + + rm -f "$FILES_TO_DELETE" "$ALL_FILES_TEMP" + + # Aggressively prune directories (including those with only dotfiles) + PRUNED_COUNT=0 + + # First, delete all dotfiles except in .git and workflow temp files (they prevent dir deletion) + find . -type f -name ".*" \ + -not -path "*/.git/*" \ + -not -name ".code-documentation-*" \ + -not -name ".doc-stage*" \ + -delete 2>/dev/null || true + + # Multiple passes to handle nested empty directories + for i in 1 2 3 4 5 6 7 8 9 10; do + PASS_COUNT=0 + while IFS= read -r empty_dir; do + if [ -d "$empty_dir" ] && [ -z "$(ls -A "$empty_dir" 2>/dev/null)" ]; then + rmdir "$empty_dir" 2>/dev/null && PASS_COUNT=$((PASS_COUNT + 1)) + fi + done < <(find . -type d -empty 2>/dev/null | grep -v "^.$" | grep -v ".git") + PRUNED_COUNT=$((PRUNED_COUNT + PASS_COUNT)) + [ "$PASS_COUNT" -eq 0 ] && break + done + + if [ "$PRUNED_COUNT" -gt 0 ]; then + echo "Pruned $PRUNED_COUNT empty directories" + fi + + # Show what's left + echo "Remaining directories (first 20):" + find . -type d -not -path "*/.git/*" -not -path "*/.git" | head -20 + echo "::endgroup::" + else + # No limit - keep all source files + cp "$ALL_SOURCE_TEMP" "$SOURCE_FILES_LIST" + fi + + # Cleanup temp file + rm -f "$ALL_SOURCE_TEMP" + + # Count remaining files + FILE_COUNT=$(count_source_files "$SOURCE_FILES_LIST") + MAIN_COUNT=$(count_main_repo_files "$SOURCE_FILES_LIST") + DEPS_COUNT=$(count_dependency_files "$SOURCE_FILES_LIST") + + echo "Source files to document: $FILE_COUNT ($MAIN_COUNT in this repository, $DEPS_COUNT in dependencies)" + + echo "::group::Source files by language" + node /tmp/ci-source.mjs breakdown "$SOURCE_FILES_LIST" | sed 's/^/ /' + echo "::endgroup::" + + echo "::group::Sample source files (first 20)" + head -20 "$SOURCE_FILES_LIST" | sed 's/^/ /' + echo "::endgroup::" + + # Output for use by subsequent steps + set_output "source_file_count" "$FILE_COUNT" + set_output "source_files_list" "$SOURCE_FILES_LIST" + + # ========================================================================= + # PULL REQUEST FIRST (progressive pull request) + # The branch and the pull request exist before any stage runs, so each + # stage commits its results the moment it finishes. A timeout loses at + # most the stage in flight. + # ========================================================================= + - name: Derive branch name + id: branch-name-early + run: | + source /tmp/workflow-helpers.sh + # Replace colons and other invalid chars with hyphens for git branch name + SAFE_RUN_ID=$(echo "$RUN_ID" | sed 's/[:]/-/g' | sed 's/[^a-zA-Z0-9._-]/-/g') + set_output "safe_run_id" "$SAFE_RUN_ID" + echo "Run ID for the branch name: $SAFE_RUN_ID" + + - name: Create docs branch + id: create-pr-branch + run: | + source /tmp/workflow-helpers.sh + + # Use sanitized run ID for branch name. Branch, PR title, status file and + # commit messages all carry the product name: ๐Ÿฆฉ Flamingo Code Documentation. + SAFE_RUN_ID="${{ steps.branch-name-early.outputs.safe_run_id }}" + BRANCH_NAME="docs/flamingo-ai-technical-writer-$SAFE_RUN_ID" + + # Configure git + git config user.name "github-actions[bot]" + git config user.email "github-actions[bot]@users.noreply.github.com" + + # Create and push empty branch + git checkout -b "$BRANCH_NAME" + + # Create initial commit to enable PR creation + echo "# ๐Ÿฆฉ Flamingo Code Documentation: Started" > .flamingo-ai-technical-writer-status.md + echo "" >> .flamingo-ai-technical-writer-status.md + echo "Run ID: $SAFE_RUN_ID" >> .flamingo-ai-technical-writer-status.md + echo "Status: In Progress" >> .flamingo-ai-technical-writer-status.md + echo "Started: $(date -u +"%Y-%m-%d %H:%M:%S UTC")" >> .flamingo-ai-technical-writer-status.md + # -f: the status file is a hidden dot-md that many target repos' .gitignore + # patterns (e.g. `.*` / `*status*`) cover โ€” without -f, `git add` fails the + # step. It's removed again in "Clean up temporary files" before the PR. + git add -f .flamingo-ai-technical-writer-status.md + git commit -m "docs: Initialize ๐Ÿฆฉ Flamingo Code Documentation run [skip ci]" + git push -u origin "$BRANCH_NAME" + + # Store branch name for later steps + set_output "branch_name" "$BRANCH_NAME" + + echo "Docs branch pushed: $BRANCH_NAME" + + - name: Ensure pull request labels exist + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + run: | + # Labels used for ๐Ÿฆฉ Flamingo Code Documentation PRs + LABELS=( + "documentation:A label for documentation-related PRs:#0075ca" + "automated:PRs created by automation/bots:#ededed" + "in-progress:Work in progress - not ready for merge:#fbca04" + ) + + for LABEL_DEF in "${LABELS[@]}"; do + LABEL_NAME=$(echo "$LABEL_DEF" | cut -d: -f1) + LABEL_DESC=$(echo "$LABEL_DEF" | cut -d: -f2) + LABEL_COLOR=$(echo "$LABEL_DEF" | cut -d: -f3 | sed 's/#//') + + # Check if label exists + if gh label list --json name --jq '.[].name' | grep -q "^${LABEL_NAME}$"; then + echo "Label exists: $LABEL_NAME" + else + echo "Creating label: $LABEL_NAME" + gh label create "$LABEL_NAME" \ + --description "$LABEL_DESC" \ + --color "$LABEL_COLOR" || true + fi + done + + - name: Open pull request + id: create-initial-pr + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }} + RUN_ID_VAR: ${{ env.RUN_ID }} + REPO_NAME: ${{ github.repository }} + DEFAULT_BRANCH: ${{ github.event.repository.default_branch }} + run: | + source /tmp/workflow-helpers.sh + + # Create PR body in a temp file (avoiding YAML parsing issues) + { + echo "๐Ÿฆฉ Flamingo Code Documentation: In Progress" + echo "" + echo "Run ID: $RUN_ID_VAR" + echo "Status: Running..." + echo "" + echo "This PR will be updated as each documentation stage completes." + echo "" + echo "Progress" + echo "- Stage 1 Inline Documentation - Starting..." + echo "- Stage 2 Architecture Analysis - Pending" + echo "- Stage 3 Tutorial Generation - Pending" + echo "- Stage 4 Repository Documentation - Pending" + echo "" + echo "Generated by ๐Ÿฆฉ Flamingo Code Documentation" + } > /tmp/pr-body.md + + # Create PR with gh CLI (works with existing branches) + PR_URL=$(gh pr create \ + --base "$DEFAULT_BRANCH" \ + --head "$BRANCH_NAME" \ + --title "[IN PROGRESS] ๐Ÿฆฉ Flamingo Code Documentation" \ + --body-file /tmp/pr-body.md \ + --label "documentation,automated,in-progress") + + # Extract PR number from URL + PR_NUMBER=$(echo "$PR_URL" | grep -oE '[0-9]+$') + + echo "Pull request #$PR_NUMBER opened: $PR_URL" + + # Set outputs for later steps + set_output "pull-request-url" "$PR_URL" + set_output "pull-request-number" "$PR_NUMBER" + + # ========================================================================= + # REMOVE DOCS THIS RUN REGENERATES + # Deletes the documentation the configured stages own before they run, so + # the result carries no orphaned files. A subtree whose stage then + # produces nothing is put back by "Restore docs no stage regenerated". + # ========================================================================= + - name: Remove docs this run regenerates + env: + DOCS_OUTPUT_PATH: ${{ env.DOCS_OUTPUT_PATH }} + REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }} + DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }} + GETTING_STARTED_OUTPUT_PATH: ${{ env.GETTING_STARTED_OUTPUT_PATH }} + DEVELOPMENT_OUTPUT_PATH: ${{ env.DEVELOPMENT_OUTPUT_PATH }} + BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }} + run: | + source /tmp/workflow-helpers.sh + + # Recorded so a stage that ends up producing nothing can restore what was + # deleted on its behalf. Without it, a skipped stage turns the pull request + # into a net DELETION of existing documentation. + PRE_CLEAN_SHA=$(git rev-parse HEAD) + echo "PRE_CLEAN_SHA=$PRE_CLEAN_SHA" >> $GITHUB_ENV + echo "Commit before removal: $PRE_CLEAN_SHA" + + # SCOPED TO THE CONFIGURED STAGES. + # + # This used to `rm -rf $DOCS_OUTPUT_PATH` unconditionally. That was safe only + # while every repo ran all four stages. With per-repo `stages`, wiping the + # whole tree when Stage 2 is disabled means the reference architecture and + # diagrams are deleted and never rebuilt โ€” the pull request becomes a net + # DELETION of existing documentation. + CLEAN_TARGETS=() + if [[ "$STAGES" == *"codewiki"* ]]; then + CLEAN_TARGETS+=("$REFERENCE_OUTPUT_PATH" "$DIAGRAMS_OUTPUT_PATH") + fi + if [[ "$STAGES" == *"tutorials"* ]]; then + CLEAN_TARGETS+=("$GETTING_STARTED_OUTPUT_PATH" "$DEVELOPMENT_OUTPUT_PATH") + fi + + # DELETE_TARGETS is what actually gets rm -rf'd; CLEAN_TARGETS is what the + # restore step keys PER-STAGE. They differ only for a full wipe: we delete + # the whole tree (so orphaned files from a previous layout โ€” e.g. synthetic + # module_N dirs left by an earlier clustering-failure run โ€” cannot survive) + # but still RECORD the per-stage subtrees, so the restore leaves each + # subtree deleted iff its OWN stage produced output. + # + # Recording the blanket DOCS_OUTPUT_PATH instead (the old behaviour) made the + # restore treat docs/ as a single unit that is "safe to leave deleted" only + # once ALL FOUR stages complete โ€” but the restore runs right after Stage 2, + # so Stage 3/4 are never 'completed' yet, and it restored the ENTIRE pre-clean + # tree every time, undoing the wipe and resurrecting the orphaned module_N docs. + DELETE_TARGETS=("${CLEAN_TARGETS[@]}") + FULL_WIPE=false + SELECTED_COUNT=$(echo "$STAGES" | tr ',' '\n' | grep -c .) + if [ "$SELECTED_COUNT" -ge "${STAGE_COUNT:-4}" ]; then + FULL_WIPE=true + DELETE_TARGETS=("$DOCS_OUTPUT_PATH") + # Record every stage-owned subtree (NOT the blanket docs/) so the restore + # keys each subtree on its own stage instead of the all-four AND. + CLEAN_TARGETS=("$REFERENCE_OUTPUT_PATH" "$DIAGRAMS_OUTPUT_PATH" "$GETTING_STARTED_OUTPUT_PATH" "$DEVELOPMENT_OUTPUT_PATH") + fi + + echo "Stages: $STAGES" + echo "Full wipe: $FULL_WIPE" + echo "Targets: ${CLEAN_TARGETS[*]:-(none)}" + + # Recorded so the restore step iterates exactly what was removed โ€” the + # stage -> subtree map has ONE home, here. + { + echo "CLEAN_TARGETS_RECORD<<__EOT__" + for t in ${CLEAN_TARGETS[@]+"${CLEAN_TARGETS[@]}"}; do echo "$t"; done + echo "__EOT__" + } >> $GITHUB_ENV + + # Count files before deletion (for reporting). Iterate DELETE_TARGETS โ€” + # the actual rm list (blanket docs/ on a full wipe, per-stage subtrees + # otherwise) โ€” not the restore-record CLEAN_TARGETS. + DELETED_FILES=0 + for target in ${DELETE_TARGETS[@]+"${DELETE_TARGETS[@]}"}; do + if [ -d "$target" ]; then + TARGET_FILES=$(find "$target" -type f | wc -l | tr -d ' ') + DELETED_FILES=$((DELETED_FILES + TARGET_FILES)) + echo "Removing $target ($TARGET_FILES files)" + rm -rf "$target" + else + echo "Skipping $target (does not exist)" + fi + done + if [ ${#DELETE_TARGETS[@]} -eq 0 ]; then + echo "No configured stage owns a docs subtree; nothing to remove" + fi + + # Recreate base directory + mkdir -p "$DOCS_OUTPUT_PATH" + + # Commit the deletion to git (so it shows in PR) + if [ "$DELETED_FILES" -gt 0 ]; then + git add -A + + # Check if there are staged changes + STAGED_COUNT=$(git diff --cached --name-only | wc -l | tr -d ' ') + + if [ "$STAGED_COUNT" -gt 0 ]; then + git commit -m "chore(docs): Clean slate - remove all documentation ($DELETED_FILES files) [skip ci]" + git push origin "$BRANCH_NAME" + echo "Committed the removal of $STAGED_COUNT files" + else + echo "Nothing to commit (the docs were already absent)" + fi + fi + + - name: Set up Node.js + uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0 + with: + node-version: '22' + + # ========================================================================= + # STAGE 0: CODE GRAPH (same build as the standalone code-graph job above) + # Tags the SOURCE branch head so the hub can render this run's + # ecosystem.md from the snapshot of the commit being documented. The hub + # promotes to `live` only when the source branch is the default branch. + # Never fatal: a graph failure costs cross-repo facts, not the docs run. + # ========================================================================= + # The command is CODE_GRAPH_INSTALL_COMMAND (lib/config/code-graph-workflow.ts). + - name: Install graph dependencies + continue-on-error: true + run: mkdir -p "$RUNNER_TEMP/code-graph-deps" && cd "$RUNNER_TEMP/code-graph-deps" && printf '%s' '{"name":"code-graph-deps","version":"1.0.0","private":true,"dependencies":{"web-tree-sitter":"0.27.0","@vscode/tree-sitter-wasm":"0.3.1","yaml":"2.9.1"}}' > package.json && printf '%s' '{"name":"code-graph-deps","version":"1.0.0","lockfileVersion":3,"requires":true,"packages":{"":{"name":"code-graph-deps","version":"1.0.0","dependencies":{"web-tree-sitter":"0.27.0","@vscode/tree-sitter-wasm":"0.3.1","yaml":"2.9.1"}},"node_modules/web-tree-sitter":{"version":"0.27.0","resolved":"https://registry.npmjs.org/web-tree-sitter/-/web-tree-sitter-0.27.0.tgz","integrity":"sha512-XK08gj6RwTMQatAG7uVRP8MunqotL/XC19vHgkSPKmELgbGPBj4ECvB8haHOUnyj6ls2B8t42UTro14zxGgAHg=="},"node_modules/@vscode/tree-sitter-wasm":{"version":"0.3.1","resolved":"https://registry.npmjs.org/@vscode/tree-sitter-wasm/-/tree-sitter-wasm-0.3.1.tgz","integrity":"sha512-RJFoomET6FajjG511fmQxeBQfU6M24a0aFZPqpid+ttIxanWf1VGytBG0UmsGjt07qmIPJS8U31D+aecuCucsQ=="},"node_modules/yaml":{"version":"2.9.1","resolved":"https://registry.npmjs.org/yaml/-/yaml-2.9.1.tgz","integrity":"sha512-3NxN8+78OdzbT7C/WjGsyfPAtJaN3FNDsWxv7Y7mcDsT/oOmgW8BpyQQFFBnvZE3j9Y2Sdz1ULFLezL7Eb2yFw=="}}}' > package-lock.json && npm ci --ignore-scripts --no-audit --no-fund && echo "CODE_GRAPH_DEPS_DIR=$RUNNER_TEMP/code-graph-deps" >> "$GITHUB_ENV" || { echo "::warning::graph dependencies failed their lockfile-enforced install; continuing without them"; rm -rf "$RUNNER_TEMP/code-graph-deps"; exit 1; } + + - name: Build and upload the code graph + id: graph + continue-on-error: true + env: + WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }} + CODE_GRAPH_DEPS_DIR: ${{ env.CODE_GRAPH_DEPS_DIR }} + GITHUB_REPOSITORY: ${{ github.repository }} + CODE_GRAPH_BRANCH: ${{ env.SOURCE_BRANCH }} + CODE_GRAPH_COMMIT_SHA: ${{ env.SOURCE_HEAD_SHA }} + run: node /tmp/code-graph-build.mjs + + # ========================================================================= + # STAGE 1: INLINE DOCS + # One hidden .md beside each source file, over the discovered file list. + # ========================================================================= + - name: Install stage 1 dependencies + if: contains(env.STAGES, 'inline-docs') + # Install generator deps in an ISOLATED tree under RUNNER_TEMP, NOT the target + # repo. npm resolves against an empty package.json here, so a target repo's own + # peer conflicts (e.g. react-accessible-accordion vs react 18) can never make this + # fail. No --legacy-peer-deps / --no-save band-aids. Generators find these via the + # NODE_PATH set on the generate step (RUNNER_TEMP/doc-orch-deps/node_modules). + run: | + mkdir -p "$RUNNER_TEMP/doc-orch-deps" && cd "$RUNNER_TEMP/doc-orch-deps" + npm init -y >/dev/null 2>&1 + npm install @anthropic-ai/sdk@0.115.0 zod@3.25.76 glob@13.0.6 + + - name: Generate inline docs (stage 1) + id: stage1 + if: contains(env.STAGES, 'inline-docs') + env: + # SECURITY: Pass secrets per-step with inline masking. + # No ANTHROPIC_API_KEY โ€” this stage calls Claude through the hub + # (/api/ci/claude), which the WEBHOOK_SECRET below authenticates. + WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }} + # Claude model SSOT โ€” see workflow env CLAUDE_MODEL block + CLAUDE_MODEL: ${{ env.CLAUDE_MODEL }} + DOCS_OUTPUT_PATH: ${{ env.DOCS_OUTPUT_PATH }} + # Stage timeout (in hours) + STAGE1_TIMEOUT_HOURS: ${{ env.STAGE1_TIMEOUT_HOURS }} + # Incremental commit + push every N generated docs (see workflow env) + STAGE1_PUSH_INTERVAL: ${{ env.STAGE1_PUSH_INTERVAL }} + BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }} + STAGE1_PUSH_MARKER: /tmp/stage1-progress-pushes + # Unified file discovery result (single source of truth) + SOURCE_FILES_LIST: ${{ steps.discover_files.outputs.source_files_list }} + SOURCE_FILE_COUNT: ${{ steps.discover_files.outputs.source_file_count }} + # NODE_PATH to find modules from /tmp/ scripts + NODE_PATH: ${{ runner.temp }}/doc-orch-deps/node_modules + # Custom AI Instructions (All Stages) + CUSTOM_REPO_INSTRUCTIONS: ${{ env.CUSTOM_INSTRUCTIONS }} + # External Repositories + EXTERNAL_REPOS: ${{ env.EXTERNAL_REPOS }} + run: | + source /tmp/workflow-helpers.sh + + echo "Source files: $SOURCE_FILE_COUNT (from $SOURCE_FILES_LIST)" + echo "Progress push: every $STAGE1_PUSH_INTERVAL generated docs, to $BRANCH_NAME" + + # Fresh marker: the generator appends one line per progress push, and the + # commit step below reads it to know work was already pushed. + rm -f "$STAGE1_PUSH_MARKER" + + # Script already downloaded to /tmp/ in setup step. run_stage records + # the outcome as stage1_status โ€” see its note in workflow-helpers.sh. + run_stage "Stage 1" "$STAGE1_TIMEOUT_HOURS" stage1_status node /tmp/generate-inline-docs.cjs + + # Count generated files (hidden .*.md files) + INLINE_DOCS=$(find . -name ".*.md" -newer .git -type f -not -path "./node_modules/*" -not -path "./.git/*" | wc -l) + set_output "stage1_files" "$INLINE_DOCS" + + # ========================================================================= + # COMMIT STAGE 1 RESULTS (progressive pull request) + # ========================================================================= + # `!= ''`, not `== 'completed'`: runs on a FAILED stage too โ€” see run_stage + # in workflow-helpers.sh (partial output is worth committing; the status + # is what reports the truth home). The build gate holds every run_stage + # commit step to this predicate. + - name: Commit and push stage 1 results + if: always() && steps.stage1.outputs.stage1_status != '' + id: commit-stage1 + env: + BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }} + STAGE_FILES: ${{ steps.stage1.outputs.stage1_files }} + STAGE1_PUSH_MARKER: /tmp/stage1-progress-pushes + run: | + source /tmp/workflow-helpers.sh + + # The generator already commits + pushes every N docs (STAGE1_PUSH_INTERVAL). + # Report how much landed that way; what's left here is the final partial batch. + if [ -s "$STAGE1_PUSH_MARKER" ]; then + PUSHED_BATCHES=$(wc -l < "$STAGE1_PUSH_MARKER" | tr -d ' ') + PUSHED_FILES=$(awk '{ sum += $1 } END { print sum + 0 }' "$STAGE1_PUSH_MARKER") + echo "Already pushed during generation: $PUSHED_FILES files in $PUSHED_BATCHES batches" + fi + + # Push any commits the generator made but could not push (transient push failure) + git push origin "HEAD:refs/heads/$BRANCH_NAME" 2>/dev/null || true + + # Stage all .md files generated by Stage 1 (hidden inline docs) + find . -name ".*.md" -type f \ + -not -path "./node_modules/*" \ + -not -path "./.git/*" \ + -exec git add -f {} \; 2>/dev/null || true + + # Check if there are changes + STAGED_COUNT=$(git diff --cached --name-only | wc -l) + + if [ "$STAGED_COUNT" -gt 0 ]; then + # Commit and push + git commit -m "docs: Stage 1 - Inline documentation ($STAGE_FILES files) [skip ci]" + git push origin "$BRANCH_NAME" + + echo "Committed and pushed $STAGED_COUNT stage 1 files" + set_output "committed" "true" + elif [ -s "$STAGE1_PUSH_MARKER" ]; then + # Everything already landed via the incremental progress pushes + echo "Nothing left to commit: every stage 1 file was pushed during generation" + set_output "committed" "true" + else + echo "::warning title=Stage 1 produced nothing::No inline docs to commit." + set_output "committed" "false" + fi + + # ========================================================================= + # ORPHANED INLINE DOCS + # Right after stage 1: removes each hidden .*.md whose source file is gone. + # ========================================================================= + # `== 'completed'` HERE IS DELIBERATE, unlike the commit steps: deleting + # "orphaned" docs after a stage that FAILED would delete docs whose + # sources were never re-examined. + - name: Remove orphaned inline docs + if: always() && steps.stage1.outputs.stage1_status == 'completed' + id: orphan-detection-inline + env: + BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }} + run: | + source /tmp/workflow-helpers.sh + + # Run orphan detection script + bash /tmp/detect-orphans.sh || true + + # Check if orphans were deleted (script writes list to /tmp/orphaned-inline-files.txt) + DELETED_FILES_LIST="/tmp/orphaned-inline-files.txt" + + if [ -f "$DELETED_FILES_LIST" ] && [ -s "$DELETED_FILES_LIST" ]; then + DELETED_COUNT=$(wc -l < "$DELETED_FILES_LIST" | tr -d ' ') + echo "Orphaned inline docs removed: $DELETED_COUNT" + + # Only add the specific files that were deleted by the script + while IFS= read -r deleted_file; do + git add "$deleted_file" 2>/dev/null || true + done < "$DELETED_FILES_LIST" + + # Verify we have staged changes + STAGED_COUNT=$(git diff --cached --name-only | wc -l | tr -d ' ') + + if [ "$STAGED_COUNT" -gt 0 ]; then + git commit -m "chore(docs): Remove $DELETED_COUNT orphaned inline files [skip ci]" + git push origin "$BRANCH_NAME" + echo "Committed $STAGED_COUNT orphan removals" + set_output "orphans_deleted" "$STAGED_COUNT" + else + echo "Nothing to commit (the files were already absent)" + set_output "orphans_deleted" "0" + fi + else + echo "No orphaned inline docs" + set_output "orphans_deleted" "0" + fi + + - name: Update pull request with stage 1 progress + if: always() && steps.commit-stage1.outputs.committed == 'true' && steps.create-initial-pr.outputs.pull-request-number + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }} + STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files }} + RUN_ID_VAR: ${{ env.RUN_ID }} + REPO_NAME: ${{ github.repository }} + run: | + source /tmp/workflow-helpers.sh + + # Update PR body + PR_BODY="## ๐Ÿฆฉ Flamingo Code Documentation: In Progress + + **Run ID:** \`$RUN_ID_VAR\` + **Status:** ๐Ÿ”„ Running... + + This PR is being updated as each documentation stage completes. + + ### Progress + - โœ… Stage 1: Inline Documentation - Completed ($STAGE1_FILES files) + - โณ Stage 2: Architecture Analysis - Running... + - โฑ๏ธ Stage 3: Tutorial Generation - Pending + - โฑ๏ธ Stage 4: Repository Documentation - Pending + + --- + ๐Ÿฆฉ Generated by [Flamingo Code Documentation](https://flamingo.run)" + + gh pr edit "$PR_NUMBER" --body "$PR_BODY" + + echo "Pull request #$PR_NUMBER updated with stage 1 progress" + + - name: Report stage 1 progress + if: always() && contains(env.STAGES, 'inline-docs') && env.HUB_BASE_URL != '' + continue-on-error: true + env: + WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }} + WORKFLOW_RUN_ID: ${{ github.run_id }} + WORKFLOW_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} + STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }} + STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }} + PR_URL: ${{ steps.create-initial-pr.outputs.pull-request-url }} + PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }} + run: | + source /tmp/workflow-helpers.sh + + CALLBACK_URL="${HUB_BASE_URL}/api/code-documentation/webhook" + echo "Reporting stage 1 (inline docs): $STAGE1_STATUS, $STAGE1_FILES files" + report_stage_progress "$CALLBACK_URL" "$WEBHOOK_SECRET" "$RUN_ID" "$REPO_ID" \ + "$WORKFLOW_RUN_ID" "$WORKFLOW_URL" "codewiki" \ + "$STAGE1_STATUS" "$STAGE1_FILES" "" "0" "" "0" "" "0" \ + "$PR_URL" "$PR_NUMBER" + + # ========================================================================= + # ECOSYSTEM FACTS โ€” derived by the hub from the code graph, never written + # by a model. Two renderings of the same live snapshot: ecosystem.md + # (committed under the reference tree and fed to the Stage 2/3/4 prompts + # as ground truth for the Dependencies sections) and the marker-delimited + # AGENTS.md block (upserted in place, idempotent). A repo with no graph + # yet is a notice, not a failure. + # ========================================================================= + - name: Fetch ecosystem facts + continue-on-error: true + env: + WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }} + REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }} + run: | + source /tmp/workflow-helpers.sh + + # Through ci-hub.mjs, the shell's way into the ONE hub transport + # (code-review-lib.mjs): it checks the destination before the secret + # leaves, keeps the secret in a header, and writes the body only on a + # 2xx. It prints the status and exits 0 whenever the hub ANSWERED โ€” a 404 + # is an answer ("no graph yet"), not a failure. + REPO_PARAM=$(printf '%s' "$GITHUB_REPOSITORY" | sed 's|/|%2F|g') + ECOSYSTEM_PATH="/api/ci/code-graph/ecosystem.md?repo=${REPO_PARAM}" + + HTTP_CODE=$(node /tmp/ci-hub.mjs get "$ECOSYSTEM_PATH" /tmp/ecosystem.md) || HTTP_CODE="000" + if [ "$HTTP_CODE" = "404" ]; then + echo "::notice title=No code graph yet::The hub has no code graph for $GITHUB_REPOSITORY yet; the docs are written without ecosystem facts." + rm -f /tmp/ecosystem.md + exit 0 + fi + if [ "$HTTP_CODE" != "200" ]; then + echo "::warning title=Ecosystem facts unavailable::The hub answered HTTP $HTTP_CODE; the Dependencies sections are written without cross-repository facts." + rm -f /tmp/ecosystem.md + exit 0 + fi + # Kept in /tmp ONLY until Stage 2 has run. The reference directory is + # CodeWiki's output directory, and CodeWiki asks "already contains + # documentation. Overwrite?" when it finds a .md file there โ€” a prompt + # a runner cannot answer, so Stage 2 aborted on every CodeWiki repo + # (CodeWiki run 35293807658). The Stage 2 commit step copies the file + # into place, after either engine has written its own output. + echo "Fetched ecosystem.md ($(wc -c < /tmp/ecosystem.md | tr -d ' ') bytes); it is copied to $REFERENCE_OUTPUT_PATH after stage 2" + + HTTP_CODE=$(node /tmp/ci-hub.mjs get "${ECOSYSTEM_PATH}&format=agents" /tmp/ecosystem-agents.md) || HTTP_CODE="000" + if [ "$HTTP_CODE" = "200" ]; then + upsert_marker_block AGENTS.md /tmp/ecosystem-agents.md + # Claude Code reads CLAUDE.md, and AGENTS.md only when a folder has NO CLAUDE.md + # (native fallback since 2.1.277). A repository that keeps a CLAUDE.md therefore gets + # ONE marker-delimited `@AGENTS.md` import, so its agents load the block above (the + # ecosystem facts and the multi-repo change-set rule). Skipped when CLAUDE.md already + # imports AGENTS.md itself; never creates a CLAUDE.md. + IMPORT_START='' + IMPORT_END='' + if [ -f CLAUDE.md ] && { grep -qF "$IMPORT_START" CLAUDE.md || ! grep -qE '(^|[[:space:]])@AGENTS\.md' CLAUDE.md; }; then + printf '%s\n' "$IMPORT_START" '@AGENTS.md' "$IMPORT_END" > /tmp/claude-agents-import.md + upsert_marker_block CLAUDE.md /tmp/claude-agents-import.md "$IMPORT_START" "$IMPORT_END" + fi + else + echo "::warning title=AGENTS.md block unavailable::The hub answered HTTP $HTTP_CODE; AGENTS.md is left untouched." + rm -f /tmp/ecosystem-agents.md + fi + + # ========================================================================= + # STAGE 2: REFERENCE DOCS + # Architecture overview, module tree and diagrams. CodeWiki where it can + # parse the primary language ("Detect primary language" decided, before + # stage 1), otherwise the Claude architecture analysis. + # ========================================================================= + + - name: Set up Python 3.12 + if: contains(env.STAGES, 'codewiki') && steps.detect_language.outputs.codewiki_supported == 'true' + uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 + with: + python-version: '3.12' + + - name: Install CodeWiki + id: codewiki_install + if: contains(env.STAGES, 'codewiki') && steps.detect_language.outputs.codewiki_supported == 'true' + continue-on-error: true + env: + CODEWIKI_MAX_FILES_PER_MODULE: ${{ env.CODEWIKI_MAX_FILES_PER_MODULE }} + CODEWIKI_REPO: ${{ env.CODEWIKI_REPO }} + run: | + # Install keyrings.alt for headless keyring support in CI environments + # Install ipython to suppress "Mermaidjs magic function not available" warning + # Install colorama for CodeWiki colored terminal output + pip install keyrings.alt ipython colorama + + # Clone CodeWiki directly (no pip caching issues) + # Fixes baked into fork: + # - retries=3 for Pydantic AI agents (prevents "Tool exceeded max retries count of 1") + # - Synthetic module creation when clustering returns 0 modules (prevents context overflow) + # - 'children' key fix for synthetic modules + # - module_tree.json path fix (commit c1dfe5c) - loads from base docs dir, not nested module dir + # See: https://github.com/flamingo-stack/CodeWiki + # Extract repo URL from CODEWIKI_REPO (strip git+ prefix and @branch/commit suffix) + REPO_URL=$(echo "$CODEWIKI_REPO" | sed 's|^git+||' | sed 's|@[^@]*$||') + REF=$(echo "$CODEWIKI_REPO" | grep -o '@[^@]*$' | sed 's|^@||' || echo "main") + echo "๐Ÿ“ฆ Cloning CodeWiki from: $REPO_URL (ref: ${REF:-main})" + rm -rf /tmp/CodeWiki + + # Clone and checkout - handle both branches and commit hashes + if [[ "${REF}" =~ ^[0-9a-f]{7,40}$ ]]; then + # Commit hash - clone full repo and checkout specific commit + git clone "$REPO_URL" /tmp/CodeWiki + cd /tmp/CodeWiki && git checkout "${REF}" && cd - + else + # Branch name - shallow clone + git clone --depth 1 --branch "${REF:-main}" "$REPO_URL" /tmp/CodeWiki + fi + + echo " Commit: $(cd /tmp/CodeWiki && git rev-parse --short HEAD)" + + # Install from local clone (reliable, no caching) + echo "๐Ÿ“ฆ Installing CodeWiki from local clone..." + pip install --no-cache-dir /tmp/CodeWiki + + source /tmp/workflow-helpers.sh + set_output "codewiki_installed" "true" + + - name: Configure CodeWiki + id: codewiki_config + if: contains(env.STAGES, 'codewiki') && steps.detect_language.outputs.codewiki_supported == 'true' && steps.codewiki_install.outputs.codewiki_installed == 'true' + continue-on-error: true + env: + # SECURITY: Pass secrets per-step with inline masking + OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + # Set keyring backend via env var (must be set before any keyring operations) + PYTHON_KEYRING_BACKEND: keyrings.alt.file.PlaintextKeyring + # Flamingo Markdown Guidelines path (needed for module import during config/validate) + FLAMINGO_MARKDOWN_GUIDELINES_PATH: /tmp/flamingo-markdown-guidelines.md + # OSS Tenant Structure: Stage 2 outputs (for clean slate deletion) + REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }} + DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }} + run: | + source /tmp/workflow-helpers.sh + + # === KEYRING CONFIGURATION FOR CI === + # CodeWiki stores API keys in system keyring. In CI (no GUI), we must: + # 1. Create keyring config to specify PlaintextKeyring backend + # 2. Create data directory for credential storage + # See: https://github.com/FSoft-AI4Code/CodeWiki - uses keyring.set_password() + + echo "๐Ÿ”‘ Setting up keyring for headless CI environment..." + + # Create keyring configuration directory and config file + mkdir -p ~/.config/python_keyring + cat > ~/.config/python_keyring/keyringrc.cfg << 'KEYRING_CFG' + [backend] + default-keyring=keyrings.alt.file.PlaintextKeyring + KEYRING_CFG + + # Ensure keyring data directory exists with proper permissions + mkdir -p ~/.local/share/python_keyring + chmod 700 ~/.local/share/python_keyring + + # Debug: Verify keyring is properly configured + echo "๐Ÿ“‹ Keyring backend verification:" + python3 -c "import keyring; print(f' Active backend: {keyring.get_keyring()}')" + + # Configure CodeWiki with separate cluster and generation providers/models + # CodeWiki calls provider APIs directly via --base-url + # Model names should match the provider's API format (no LiteLLM prefix needed) + # OpenAI: gpt-4o, gpt-4-turbo, gpt-4o-mini + # Provider ids come from MODEL_METADATA in lib/constants/ai-models.ts + # See: https://github.com/FSoft-AI4Code/CodeWiki + + echo "๐Ÿ”ง Configuring CodeWiki..." + echo " Cluster (Phase 2): $CODEWIKI_CLUSTER_PROVIDER / $CODEWIKI_CLUSTER_MODEL" + echo " Generation (Phase 3+): $CODEWIKI_GENERATION_PROVIDER / $CODEWIKI_GENERATION_MODEL" + + # Source helper functions for configuration + source /tmp/workflow-helpers.sh + + # Determine API keys for each provider (cluster, generation/main, fallback) + # Each provider can use a different AI service (OpenAI, Anthropic, etc.) + if [ "$CODEWIKI_CLUSTER_PROVIDER" = "anthropic" ]; then + CLUSTER_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}" + echo " Cluster: Using Anthropic API key" + else + CLUSTER_API_KEY="${{ secrets.OPENAI_API_KEY }}" + echo " Cluster: Using OpenAI API key" + fi + + if [ "$CODEWIKI_GENERATION_PROVIDER" = "anthropic" ]; then + MAIN_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}" + echo " Generation: Using Anthropic API key" + else + MAIN_API_KEY="${{ secrets.OPENAI_API_KEY }}" + echo " Generation: Using OpenAI API key" + fi + + if [ "$CODEWIKI_FALLBACK_PROVIDER" = "anthropic" ]; then + FALLBACK_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}" + echo " Fallback: Using Anthropic API key" + else + FALLBACK_API_KEY="${{ secrets.OPENAI_API_KEY }}" + echo " Fallback: Using OpenAI API key" + fi + + # Configure CodeWiki from CODEWIKI_CONFIG_JSON (single extractor: configure_codewiki_from_json) + # Pass per-provider API keys for mixed provider configurations + configure_codewiki_from_json "$CODEWIKI_CONFIG_JSON" "$CLUSTER_API_KEY" "$MAIN_API_KEY" "$FALLBACK_API_KEY" + + if [ $? -ne 0 ]; then + echo "โŒ CodeWiki configuration failed" + exit 1 + fi + + # Set environment variables for backward compatibility with run-codewiki-analysis.sh + export MAIN_MODEL="$CODEWIKI_GENERATION_MODEL" + export FALLBACK_MODEL_1="$CODEWIKI_FALLBACK_MODEL" + if [ "$PRIMARY_PROVIDER" = "anthropic" ]; then + export ANTHROPIC_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}" + else + export OPENAI_API_KEY="${{ secrets.OPENAI_API_KEY }}" + fi + + # Verify configuration was saved + echo "" + echo "๐Ÿ“‹ CodeWiki configuration:" + python -m codewiki config show + + echo "" + echo "โœ… Validating configuration..." + python -m codewiki config validate + + echo "" + set_output "codewiki_configured" "true" + + - name: Generate reference docs with CodeWiki (stage 2) + id: stage2 + if: contains(env.STAGES, 'codewiki') && steps.detect_language.outputs.codewiki_supported == 'true' && steps.codewiki_config.outputs.codewiki_configured == 'true' + continue-on-error: false + env: + # Keyring backend for CI (must match config step) + PYTHON_KEYRING_BACKEND: keyrings.alt.file.PlaintextKeyring + # API keys for both providers (CodeWiki will use the one configured) + OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + # OSS Tenant Structure: Stage 2 outputs + REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }} + DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }} + # Stage timeout + STAGE2_TIMEOUT_HOURS: ${{ env.STAGE2_TIMEOUT_HOURS }} + # CodeWiki JSON configuration (required for unified function) + CODEWIKI_CONFIG_JSON: ${{ env.CODEWIKI_CONFIG_JSON }} + # CodeWiki model configuration (cluster, generation, fallback) + CODEWIKI_CLUSTER_PROVIDER: ${{ env.CODEWIKI_CLUSTER_PROVIDER }} + CODEWIKI_CLUSTER_MODEL: ${{ env.CODEWIKI_CLUSTER_MODEL }} + CODEWIKI_CLUSTER_MAX_TOKENS: ${{ env.CODEWIKI_CLUSTER_MAX_TOKENS }} + CODEWIKI_CLUSTER_MAX_TOKEN_FIELD: ${{ env.CODEWIKI_CLUSTER_MAX_TOKEN_FIELD }} + CODEWIKI_GENERATION_PROVIDER: ${{ env.CODEWIKI_GENERATION_PROVIDER }} + CODEWIKI_GENERATION_MODEL: ${{ env.CODEWIKI_GENERATION_MODEL }} + CODEWIKI_GENERATION_MAX_TOKENS: ${{ env.CODEWIKI_GENERATION_MAX_TOKENS }} + CODEWIKI_GENERATION_MAX_TOKEN_FIELD: ${{ env.CODEWIKI_GENERATION_MAX_TOKEN_FIELD }} + CODEWIKI_FALLBACK_PROVIDER: ${{ env.CODEWIKI_FALLBACK_PROVIDER }} + CODEWIKI_FALLBACK_MODEL: ${{ env.CODEWIKI_FALLBACK_MODEL }} + CODEWIKI_FALLBACK_MAX_TOKENS: ${{ env.CODEWIKI_FALLBACK_MAX_TOKENS }} + CODEWIKI_FALLBACK_MAX_TOKEN_FIELD: ${{ env.CODEWIKI_FALLBACK_MAX_TOKEN_FIELD }} + CODEWIKI_MAX_DEPTH: ${{ env.CODEWIKI_MAX_DEPTH }} + # Flamingo Markdown Guidelines path for CodeWiki prompts + FLAMINGO_MARKDOWN_GUIDELINES_PATH: /tmp/flamingo-markdown-guidelines.md + # Markdown Validation Rules (injected into all prompts) + VALIDATION_RULES_PATH: ${{ env.VALIDATION_RULES_PATH }} + # Custom AI Instructions (All Stages) + CUSTOM_REPO_INSTRUCTIONS: ${{ env.CUSTOM_INSTRUCTIONS }} + # External Repositories + EXTERNAL_REPOS: ${{ env.EXTERNAL_REPOS }} + # Dependencies (for CodeWiki multi-path support) + DEPENDENCIES: ${{ env.DEPENDENCIES }} + run: | + # Determine per-provider API keys (same logic as Configure step) + if [ "$CODEWIKI_CLUSTER_PROVIDER" = "anthropic" ]; then + export CLUSTER_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}" + else + export CLUSTER_API_KEY="${{ secrets.OPENAI_API_KEY }}" + fi + + if [ "$CODEWIKI_GENERATION_PROVIDER" = "anthropic" ]; then + export MAIN_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}" + else + export MAIN_API_KEY="${{ secrets.OPENAI_API_KEY }}" + fi + + if [ "$CODEWIKI_FALLBACK_PROVIDER" = "anthropic" ]; then + export FALLBACK_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}" + else + export FALLBACK_API_KEY="${{ secrets.OPENAI_API_KEY }}" + fi + + # Verify dependencies directory before CodeWiki runs + echo "" + echo "๐Ÿ” Pre-CodeWiki Dependency Verification:" + echo " Current directory: $(pwd)" + echo " Absolute path: $(realpath .)" + echo "" + + if [ -d "./deps" ]; then + echo " โœ… ./deps EXISTS" + echo " Contents: $(ls -1 ./deps 2>/dev/null | wc -l) repositories" + ls -la ./deps 2>/dev/null | head -5 + else + echo " โŒ ./deps NOT FOUND" + fi + + if [ -d "../deps" ]; then + echo " โœ… ../deps EXISTS" + echo " Absolute: $(realpath ../deps)" + echo " Contents: $(ls -1 ../deps 2>/dev/null | wc -l) repositories" + ls -la ../deps 2>/dev/null | head -5 + else + echo " โŒ ../deps NOT FOUND" + fi + + echo " DEPENDENCIES env: ${DEPENDENCIES:-}" + echo "" + + # Run externalized CodeWiki analysis script + /tmp/run-codewiki-analysis.sh + + # The alternative for any language CodeWiki cannot parse. + - name: Generate reference docs with Claude (stage 2) + id: stage2_alt + if: contains(env.STAGES, 'codewiki') && steps.detect_language.outputs.codewiki_supported == 'false' + env: + # No ANTHROPIC_API_KEY โ€” this stage calls Claude through the hub + # (/api/ci/claude), which the WEBHOOK_SECRET below authenticates. + WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }} + # SSOT โ€” see workflow env CLAUDE_MODEL block + CLAUDE_MODEL: ${{ env.CLAUDE_MODEL }} + PRIMARY_LANGUAGE: ${{ steps.detect_language.outputs.primary_language }} + # OSS Tenant Structure: Stage 2 outputs + REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }} + DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }} + # Unified file discovery result (single source of truth) + SOURCE_FILES_LIST: ${{ steps.discover_files.outputs.source_files_list }} + SOURCE_FILE_COUNT: ${{ steps.discover_files.outputs.source_file_count }} + # Custom AI Instructions (All Stages) + CUSTOM_REPO_INSTRUCTIONS: ${{ env.CUSTOM_INSTRUCTIONS }} + # External Repositories + EXTERNAL_REPOS: ${{ env.EXTERNAL_REPOS }} + # Dependencies (for consistency with other stages) + DEPENDENCIES: ${{ env.DEPENDENCIES }} + run: | + # Run externalized Claude architecture analysis script + /tmp/run-claude-architecture-analysis.sh + + # ========================================================================= + # RESTORE DOCS NO STAGE REGENERATED + # The docs removal took the reference/diagrams trees because `codewiki` was in + # STAGES. If neither Stage-2 variant then completed โ€” a repo with no + # discoverable source, an install failure, a skip โ€” the deletion would be the + # only Stage-2 change in the pull request, i.e. a net removal of documentation + # nobody asked to remove. Put it back. + # ========================================================================= + - name: Restore docs no stage regenerated + if: always() + env: + BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }} + STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status }} + STAGE2_STATUS: ${{ steps.stage2.outputs.stage2_status }} + STAGE2_ALT_STATUS: ${{ steps.stage2_alt.outputs.stage2_status }} + run: | + # The docs removal deleted whatever the configured stages own, and committed that + # deletion. Any owned subtree whose stage then produced nothing must be put + # back โ€” otherwise the pull request is a net REMOVAL of documentation nobody + # asked to remove. Iterates the exact list the removal recorded, so the + # stage -> subtree map has one home. + if [ -z "$CLEAN_TARGETS_RECORD" ]; then + echo "Nothing was removed; nothing to restore" + exit 0 + fi + + STAGE2_OK=false + [ "$STAGE2_STATUS" = "completed" ] && STAGE2_OK=true + [ "$STAGE2_ALT_STATUS" = "completed" ] && STAGE2_OK=true + + # CodeWiki leaves multi-GB scratch behind on a failed run; never let it near + # the index. + rm -rf "$REFERENCE_OUTPUT_PATH/temp" "$DIAGRAMS_OUTPUT_PATH/temp" 2>/dev/null || true + + RESTORED=0 + while IFS= read -r target; do + [ -n "$target" ] || continue + # A target is safe to leave deleted only if something regenerated it. + case "$target" in + "$REFERENCE_OUTPUT_PATH"|"$DIAGRAMS_OUTPUT_PATH") + [ "$STAGE2_OK" = true ] && continue ;; + "$GETTING_STARTED_OUTPUT_PATH"|"$DEVELOPMENT_OUTPUT_PATH") + [ "$STAGE3_STATUS" = "completed" ] && continue ;; + "$DOCS_OUTPUT_PATH") + # Full wipe: only fully safe when every stage delivered. + if [ "$STAGE1_STATUS" = "completed" ] && [ "$STAGE2_OK" = true ] && \ + [ "$STAGE3_STATUS" = "completed" ] && [ "$STAGE4_STATUS" = "completed" ]; then + continue + fi ;; + esac + if git checkout "$PRE_CLEAN_SHA" -- "$target" 2>/dev/null; then + echo "::notice title=Docs restored::$target restored from $PRE_CLEAN_SHA; no stage regenerated it." + RESTORED=1 + fi + done <<< "$CLEAN_TARGETS_RECORD" + + # `git checkout -- ` already stages exactly those paths. Deliberately NO + # `git add -A`: at this point the workspace holds npm install output from + # Stage 1 and, on a failed CodeWiki run, its scratch trees. + if [ "$RESTORED" -eq 1 ] && [ -n "$(git diff --cached --name-only)" ]; then + git commit -m "chore(docs): restore documentation no stage regenerated [skip ci]" + git push origin "$BRANCH_NAME" + echo "Restore committed and pushed" + else + echo "Nothing to restore" + fi + + # ========================================================================= + # COMMIT STAGE 2 RESULTS (progressive pull request) + # ========================================================================= + - name: Commit and push stage 2 results + if: always() && (steps.stage2.outputs.stage2_status == 'completed' || steps.stage2_alt.outputs.stage2_status == 'completed') + id: commit-stage2 + env: + BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }} + STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files }} + REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }} + DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }} + run: | + source /tmp/workflow-helpers.sh + + # Remove CodeWiki scratch from every output directory + rm -rf "$REFERENCE_OUTPUT_PATH/temp" 2>/dev/null || echo "::warning::Could not remove $REFERENCE_OUTPUT_PATH/temp" + rm -rf "$DIAGRAMS_OUTPUT_PATH/temp" 2>/dev/null || echo "::warning::Could not remove $DIAGRAMS_OUTPUT_PATH/temp" + + # Verify cleanup + if [ -d "$REFERENCE_OUTPUT_PATH/temp" ]; then + echo "::error title=Scratch not removed::$REFERENCE_OUTPUT_PATH/temp is still present; refusing to commit it." + ls -la "$REFERENCE_OUTPUT_PATH/temp" + exit 1 + fi + + # .gitignore in each output directory keeps CodeWiki scratch out of commits + for output_dir in "$REFERENCE_OUTPUT_PATH" "$DIAGRAMS_OUTPUT_PATH"; do + if [ -d "$output_dir" ]; then + { + echo "# CodeWiki temp files (dependency graphs can be 7GB+)" + echo "temp/" + echo "dependency_graphs/" + echo "" + echo "# JSON intermediate files (except schema/config)" + echo "*.json" + echo "!*-schema.json" + echo "!*-config.json" + } > "$output_dir/.gitignore" + echo "Wrote $output_dir/.gitignore" + fi + done + + echo "::group::Stage 2 output before staging" + echo "Reference directory ($REFERENCE_OUTPUT_PATH):" + if [ -d "$REFERENCE_OUTPUT_PATH" ]; then + find "$REFERENCE_OUTPUT_PATH" -type f \( -name "*.md" -o -name "*.mmd" -o -name ".gitignore" \) | head -20 + FILE_COUNT=$(find "$REFERENCE_OUTPUT_PATH" -type f \( -name "*.md" -o -name "*.mmd" \) | wc -l | tr -d ' ') + echo ".md/.mmd files: $FILE_COUNT" + else + echo "(directory does not exist)" + fi + echo "Diagrams directory ($DIAGRAMS_OUTPUT_PATH):" + if [ -d "$DIAGRAMS_OUTPUT_PATH" ]; then + find "$DIAGRAMS_OUTPUT_PATH" -type f \( -name "*.md" -o -name "*.mmd" -o -name ".gitignore" \) | head -20 + FILE_COUNT=$(find "$DIAGRAMS_OUTPUT_PATH" -type f \( -name "*.md" -o -name "*.mmd" \) | wc -l | tr -d ' ') + echo ".md/.mmd files: $FILE_COUNT" + else + echo "(directory does not exist)" + fi + echo "::endgroup::" + + # The hub-rendered ecosystem.md joins the reference directory only NOW, + # after the engine has run: placed earlier it makes CodeWiki prompt for + # an overwrite (see "Fetch ecosystem facts"). + if [ -f /tmp/ecosystem.md ]; then + mkdir -p "$REFERENCE_OUTPUT_PATH" + cp /tmp/ecosystem.md "$REFERENCE_OUTPUT_PATH/ecosystem.md" + echo "Added ecosystem.md to $REFERENCE_OUTPUT_PATH/" + fi + + # Stage all .md, .mmd, and .gitignore files from Stage 2 output directories + # CRITICAL: Use git add on full paths to preserve nested directory structure + # This ensures Backend/Authentication/JWT/JWT.md keeps its full path in git + + # Add only .md, .mmd, .gitignore, and allowed JSON files (*-schema.json, *-config.json) + # This excludes CodeWiki intermediate files: module_tree.json, first_module_tree.json, metadata.json + if [ -d "$REFERENCE_OUTPUT_PATH" ]; then + find "$REFERENCE_OUTPUT_PATH" -type f \( \ + -name "*.md" -o -name "*.mmd" -o -name ".gitignore" \ + -o -name "*-schema.json" -o -name "*-config.json" \ + \) -exec git add {} \; 2>/dev/null || true + echo "Staged .md/.mmd/.gitignore/*-schema.json/*-config.json under $REFERENCE_OUTPUT_PATH/" + fi + + if [ -d "$DIAGRAMS_OUTPUT_PATH" ]; then + find "$DIAGRAMS_OUTPUT_PATH" -type f \( \ + -name "*.md" -o -name "*.mmd" -o -name ".gitignore" \ + -o -name "*-schema.json" -o -name "*-config.json" \ + \) -exec git add {} \; 2>/dev/null || true + echo "Staged .md/.mmd/.gitignore/*-schema.json/*-config.json under $DIAGRAMS_OUTPUT_PATH/" + fi + + # AGENTS.md carries the hub-rendered ecosystem block ("Fetch ecosystem + # facts" upserts it just before this stage). It sits at the repository + # root, outside every output directory staged above, so it is staged by + # name: the first production run wrote the block and no commit ever + # picked the file up (openframe-cli#383). + if [ -f AGENTS.md ]; then + git add -f AGENTS.md + echo "Staged AGENTS.md (ecosystem block)" + fi + # CLAUDE.md carries the `@AGENTS.md` import the same step keeps (only when it changed). + if [ -f CLAUDE.md ] && ! git diff --quiet -- CLAUDE.md; then + git add -f CLAUDE.md + echo " Added CLAUDE.md (@AGENTS.md import)" + fi + + # Check if there are changes + STAGED_COUNT=$(git diff --cached --name-only | wc -l) + echo "Staged files: $STAGED_COUNT" + + if [ "$STAGED_COUNT" -gt 0 ]; then + echo "::group::Staged stage 2 files (first 30)" + git diff --cached --name-only | head -30 + echo "::endgroup::" + fi + + if [ "$STAGED_COUNT" -gt 0 ]; then + # Commit and push + git commit -m "docs: Stage 2 - Architecture analysis ($STAGE2_FILES files) [skip ci]" + git push origin "$BRANCH_NAME" + + echo "Committed and pushed $STAGED_COUNT stage 2 files" + set_output "committed" "true" + else + echo "::error title=Stage 2 output not staged::Stage 2 reported $STAGE2_FILES files but none were staged; the files were generated outside the reference and diagrams directories, or not at all." + set_output "committed" "false" + fi + + - name: Update pull request with stage 2 progress + if: always() && steps.commit-stage2.outputs.committed == 'true' && steps.create-initial-pr.outputs.pull-request-number + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }} + STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }} + STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }} + RUN_ID_VAR: ${{ env.RUN_ID }} + REPO_NAME: ${{ github.repository }} + run: | + source /tmp/workflow-helpers.sh + + # Update PR body + PR_BODY="## ๐Ÿฆฉ Flamingo Code Documentation: In Progress + + **Run ID:** \`$RUN_ID_VAR\` + **Status:** ๐Ÿ”„ Running... + + This PR is being updated as each documentation stage completes. + + ### Progress + - โœ… Stage 1: Inline Documentation - Completed ($STAGE1_FILES files) + - โœ… Stage 2: Architecture Analysis - Completed ($STAGE2_FILES files) + - โณ Stage 3: Tutorial Generation - Running... + - โฑ๏ธ Stage 4: Repository Documentation - Pending + + --- + ๐Ÿฆฉ Generated by [Flamingo Code Documentation](https://flamingo.run)" + + gh pr edit "$PR_NUMBER" --body "$PR_BODY" + + echo "Pull request #$PR_NUMBER updated with stage 2 progress" + + - name: Report stage 2 progress + if: always() && contains(env.STAGES, 'codewiki') && env.HUB_BASE_URL != '' + continue-on-error: true + env: + WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }} + WORKFLOW_RUN_ID: ${{ github.run_id }} + WORKFLOW_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} + STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }} + STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }} + # Use outputs from either CodeWiki (stage2) or Claude alternative (stage2_alt) + STAGE2_STATUS: ${{ steps.stage2.outputs.stage2_status || steps.stage2_alt.outputs.stage2_status || 'skipped' }} + STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }} + PR_URL: ${{ steps.create-initial-pr.outputs.pull-request-url }} + PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }} + run: | + source /tmp/workflow-helpers.sh + + CALLBACK_URL="${HUB_BASE_URL}/api/code-documentation/webhook" + echo "Reporting stage 2 (reference docs): $STAGE2_STATUS, $STAGE2_FILES files" + report_stage_progress "$CALLBACK_URL" "$WEBHOOK_SECRET" "$RUN_ID" "$REPO_ID" \ + "$WORKFLOW_RUN_ID" "$WORKFLOW_URL" "tutorials" \ + "$STAGE1_STATUS" "$STAGE1_FILES" "$STAGE2_STATUS" "$STAGE2_FILES" "" "0" "" "0" \ + "$PR_URL" "$PR_NUMBER" + + # ========================================================================= + # STAGE 3: TUTORIALS + # Getting-started guides and how-to tutorials, written by the Code + # Documentation lane (code-documentation-lib.mjs): one hub call per + # tutorial with the forced `emit_document` tool and, when the repository's + # "Graph lookups" switch is on, the hub's read tools (the code graph and + # the rules, scoped to what THIS repository may see). No model key here. + # Generates 4 tutorials: user/getting-started, user/common-use-cases, + # dev/getting-started-dev, dev/architecture-overview-dev + # ========================================================================= + - name: Install tutorial and repository docs dependencies + # Stage 3 AND Stage 4 (generate-repo-docs.cjs) share this tree. Gated + # on either: a repository configured with inline-docs + repo-docs and no + # tutorials reached Stage 4 with no dependencies at all, run 35302244804. + if: contains(env.STAGES, 'tutorials') || contains(env.STAGES, 'repo-docs') + # Isolated deps tree (see Stage 1) - no reconciliation with the target repo. + # No model SDK: both stages call Claude through the hub. + run: | + mkdir -p "$RUNNER_TEMP/doc-orch-deps" && cd "$RUNNER_TEMP/doc-orch-deps" + npm init -y >/dev/null 2>&1 + npm install zod@3.25.76 glob@13.0.6 + + - name: Generate tutorials (stage 3) + id: stage3 + if: contains(env.STAGES, 'tutorials') + env: + # No ANTHROPIC_API_KEY โ€” this stage calls Claude through the hub + # (/api/ci/claude), which the WEBHOOK_SECRET below authenticates. + WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }} + # Pass through output paths from workflow env (OSS Tenant Structure) + DOCS_OUTPUT_PATH: ${{ env.DOCS_OUTPUT_PATH }} + # Stage 2 outputs (for context) + REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }} + DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }} + # Stage 3 outputs + GETTING_STARTED_OUTPUT_PATH: ${{ env.GETTING_STARTED_OUTPUT_PATH }} + DEVELOPMENT_OUTPUT_PATH: ${{ env.DEVELOPMENT_OUTPUT_PATH }} + # Claude model SSOT โ€” see workflow env CLAUDE_MODEL block + CLAUDE_MODEL: ${{ env.CLAUDE_MODEL }} + # Stage timeout + STAGE3_TIMEOUT_HOURS: ${{ env.STAGE3_TIMEOUT_HOURS }} + # Unified file discovery result (same files as Stage 1 and 2) + SOURCE_FILES_LIST: ${{ steps.discover_files.outputs.source_files_list }} + # NODE_PATH to find modules from /tmp/ scripts + NODE_PATH: ${{ runner.temp }}/doc-orch-deps/node_modules + # YouTube Integration - SECURITY: API key from secrets, NOT dispatch payload + YOUTUBE_ENABLED: ${{ env.YOUTUBE_ENABLED }} + YOUTUBE_CHANNELS: ${{ env.YOUTUBE_CHANNELS }} + YOUTUBE_API_KEY: ${{ secrets.YOUTUBE_API_KEY }} + # Markdown Validation Rules (injected into prompts) + VALIDATION_RULES_PATH: ${{ env.VALIDATION_RULES_PATH }} + # Flamingo Markdown Guidelines (optional) + GUIDELINES_PATH: ${{ env.GUIDELINES_PATH }} + # Stage 3 tracking files (configurable paths) + STAGE3_FILES_TRACKER: ${{ env.STAGE3_FILES_TRACKER }} + STAGE3_STATS_FILE: ${{ env.STAGE3_STATS_FILE }} + # Custom AI Instructions (All Stages) + CUSTOM_REPO_INSTRUCTIONS: ${{ env.CUSTOM_INSTRUCTIONS }} + # Analysis Exclusions + EXCLUDED_PATHS: ${{ env.EXCLUDED_PATHS }} + # External Repositories + EXTERNAL_REPOS: ${{ env.EXTERNAL_REPOS }} + run: | + source /tmp/workflow-helpers.sh + + # Script already downloaded to /tmp/ in setup step. run_stage records + # the outcome as stage3_status โ€” see its note in workflow-helpers.sh. + run_stage "Stage 3" "$STAGE3_TIMEOUT_HOURS" stage3_status node /tmp/generate-tutorials-voltagent.cjs + + # Count files from both OSS Tenant Structure directories + GETTING_STARTED_FILES=$(count_markdown_files "${GETTING_STARTED_OUTPUT_PATH}") + DEVELOPMENT_FILES=$(count_markdown_files "${DEVELOPMENT_OUTPUT_PATH}") + TUTORIAL_FILES=$((GETTING_STARTED_FILES + DEVELOPMENT_FILES)) + echo "Tutorials written: $TUTORIAL_FILES ($GETTING_STARTED_FILES getting started, $DEVELOPMENT_FILES development)" + set_output "stage3_files" "$TUTORIAL_FILES" + + # ========================================================================= + # COMMIT STAGE 3 RESULTS (progressive pull request) + # ========================================================================= + # `!= ''` โ€” the run_stage commit rule, stated once at Stage 1. + - name: Commit and push stage 3 results + if: always() && steps.stage3.outputs.stage3_status != '' + id: commit-stage3 + env: + BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }} + STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files }} + GETTING_STARTED_OUTPUT_PATH: ${{ env.GETTING_STARTED_OUTPUT_PATH }} + DEVELOPMENT_OUTPUT_PATH: ${{ env.DEVELOPMENT_OUTPUT_PATH }} + run: | + source /tmp/workflow-helpers.sh + + # .gitignore in each tutorial output directory + for output_dir in "$GETTING_STARTED_OUTPUT_PATH" "$DEVELOPMENT_OUTPUT_PATH"; do + if [ -d "$output_dir" ]; then + { + echo "# VoltAgent temp files" + echo "temp/" + echo "" + echo "# JSON intermediate files (except schema/config)" + echo "*.json" + echo "!*-schema.json" + echo "!*-config.json" + } > "$output_dir/.gitignore" + echo "Wrote $output_dir/.gitignore" + fi + done + + # Stage all .md and .gitignore files from Stage 3 output directories + find "$GETTING_STARTED_OUTPUT_PATH" "$DEVELOPMENT_OUTPUT_PATH" -type f \( -name "*.md" -o -name ".gitignore" \) \ + -exec git add -f {} \; 2>/dev/null || true + + # Check if there are changes + STAGED_COUNT=$(git diff --cached --name-only | wc -l) + + if [ "$STAGED_COUNT" -gt 0 ]; then + # Commit and push + git commit -m "docs: Stage 3 - Tutorial generation ($STAGE3_FILES files) [skip ci]" + git push origin "$BRANCH_NAME" + + echo "Committed and pushed $STAGED_COUNT stage 3 files" + set_output "committed" "true" + else + echo "::warning title=Stage 3 produced nothing::No tutorials to commit." + set_output "committed" "false" + fi + + - name: Update pull request with stage 3 progress + if: always() && steps.commit-stage3.outputs.committed == 'true' && steps.create-initial-pr.outputs.pull-request-number + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }} + STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }} + STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }} + STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || 0 }} + RUN_ID_VAR: ${{ env.RUN_ID }} + REPO_NAME: ${{ github.repository }} + run: | + source /tmp/workflow-helpers.sh + + # Update PR body + PR_BODY="## ๐Ÿฆฉ Flamingo Code Documentation: In Progress + + **Run ID:** \`$RUN_ID_VAR\` + **Status:** ๐Ÿ”„ Running... + + This PR is being updated as each documentation stage completes. + + ### Progress + - โœ… Stage 1: Inline Documentation - Completed ($STAGE1_FILES files) + - โœ… Stage 2: Architecture Analysis - Completed ($STAGE2_FILES files) + - โœ… Stage 3: Tutorial Generation - Completed ($STAGE3_FILES files) + - โณ Stage 4: Repository Documentation - Running... + + --- + ๐Ÿฆฉ Generated by [Flamingo Code Documentation](https://flamingo.run)" + + gh pr edit "$PR_NUMBER" --body "$PR_BODY" + + echo "Pull request #$PR_NUMBER updated with stage 3 progress" + + - name: Report stage 3 progress + if: always() && contains(env.STAGES, 'tutorials') && env.HUB_BASE_URL != '' + continue-on-error: true + env: + WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }} + WORKFLOW_RUN_ID: ${{ github.run_id }} + WORKFLOW_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} + STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }} + STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }} + STAGE2_STATUS: ${{ steps.stage2.outputs.stage2_status || steps.stage2_alt.outputs.stage2_status || 'skipped' }} + STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }} + STAGE3_STATUS: ${{ steps.stage3.outputs.stage3_status || 'skipped' }} + STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || 0 }} + PR_URL: ${{ steps.create-initial-pr.outputs.pull-request-url }} + PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }} + run: | + source /tmp/workflow-helpers.sh + + CALLBACK_URL="${HUB_BASE_URL}/api/code-documentation/webhook" + echo "Reporting stage 3 (tutorials): $STAGE3_STATUS, $STAGE3_FILES files" + report_stage_progress "$CALLBACK_URL" "$WEBHOOK_SECRET" "$RUN_ID" "$REPO_ID" \ + "$WORKFLOW_RUN_ID" "$WORKFLOW_URL" "repo-docs" \ + "$STAGE1_STATUS" "$STAGE1_FILES" "$STAGE2_STATUS" "$STAGE2_FILES" "$STAGE3_STATUS" "$STAGE3_FILES" "" "0" \ + "$PR_URL" "$PR_NUMBER" + + # ========================================================================= + # STAGE 4: REPOSITORY DOCS + # Copies LICENSE.md, SECURITY.md from template repo + # Generates/updates README.md, CONTRIBUTING.md and the docs index through + # the Code Documentation lane (one hub call per document, no model key) + # ========================================================================= + - name: Generate repository docs (stage 4) + id: stage4 + if: contains(env.STAGES, 'repo-docs') + env: + # No ANTHROPIC_API_KEY โ€” this stage calls Claude through the hub. + WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }} + TEMPLATE_REPO: ${{ env.TEMPLATE_REPO }} + TEMPLATE_BRANCH: ${{ env.TEMPLATE_BRANCH }} + DOCS_OUTPUT_PATH: ${{ env.DOCS_OUTPUT_PATH }} + # OSS Tenant Structure: All output paths for docs/README.md navigation + REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }} + DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }} + GETTING_STARTED_OUTPUT_PATH: ${{ env.GETTING_STARTED_OUTPUT_PATH }} + DEVELOPMENT_OUTPUT_PATH: ${{ env.DEVELOPMENT_OUTPUT_PATH }} + # Claude model SSOT โ€” see workflow env CLAUDE_MODEL block + CLAUDE_MODEL: ${{ env.CLAUDE_MODEL }} + STAGE4_TIMEOUT_HOURS: ${{ env.STAGE4_TIMEOUT_HOURS }} + # NODE_PATH to find modules from /tmp/ scripts + NODE_PATH: ${{ runner.temp }}/doc-orch-deps/node_modules + # YouTube Integration - SECURITY: API key from secrets, NOT dispatch payload + YOUTUBE_ENABLED: ${{ env.YOUTUBE_ENABLED }} + YOUTUBE_CHANNELS: ${{ env.YOUTUBE_CHANNELS }} + YOUTUBE_API_KEY: ${{ secrets.YOUTUBE_API_KEY }} + # Markdown Validation Rules (injected into prompts) + VALIDATION_RULES_PATH: ${{ env.VALIDATION_RULES_PATH }} + # Flamingo Markdown Guidelines (optional) + GUIDELINES_PATH: ${{ env.GUIDELINES_PATH }} + # Stage 4 tracking files (configurable paths) + STAGE4_FILES_TRACKER: ${{ env.STAGE4_FILES_TRACKER }} + # Custom AI Instructions (All Stages) + CUSTOM_REPO_INSTRUCTIONS: ${{ env.CUSTOM_INSTRUCTIONS }} + # Analysis Exclusions + EXCLUDED_PATHS: ${{ env.EXCLUDED_PATHS }} + # External Repositories + EXTERNAL_REPOS: ${{ env.EXTERNAL_REPOS }} + # README Branding + README_LOGO_DARK: ${{ env.README_LOGO_DARK }} + README_LOGO_LIGHT: ${{ env.README_LOGO_LIGHT }} + README_LOGO_ALT: ${{ env.README_LOGO_ALT }} + run: | + source /tmp/workflow-helpers.sh + + echo "Template repository: $TEMPLATE_REPO@$TEMPLATE_BRANCH" + + RAW_URL="https://raw.githubusercontent.com/$TEMPLATE_REPO/$TEMPLATE_BRANCH" + + # 1. LICENSE.md and SECURITY.md from the template repository + if curl -fsSL "$RAW_URL/LICENSE.md" -o LICENSE.md 2>/dev/null; then + echo "Copied LICENSE.md from the template repository" + else + echo "::notice title=LICENSE.md not copied::The template repository $TEMPLATE_REPO@$TEMPLATE_BRANCH has no LICENSE.md." + fi + + if curl -fsSL "$RAW_URL/SECURITY.md" -o SECURITY.md 2>/dev/null; then + echo "Copied SECURITY.md from the template repository" + else + echo "::notice title=SECURITY.md not copied::The template repository $TEMPLATE_REPO@$TEMPLATE_BRANCH has no SECURITY.md." + fi + + # 2. The existing README is context for the new one + if [ -f "README.md" ]; then + README_SIZE=$(wc -c < README.md | tr -d ' ') + echo "Existing README.md: $README_SIZE bytes (used as context)" + else + echo "Existing README.md: none" + fi + + # 3. README, CONTRIBUTING and the docs index, one hub call each. + # Script already downloaded to /tmp/ in "Download pipeline scripts" + # run_stage records the outcome as stage4_status โ€” see its note in + # workflow-helpers.sh. The file count below is REPORTING, not a + # status: inferring "completed" from it meant a crashed run that left + # a previous commit's README standing reported success. + run_stage "Stage 4" "$STAGE4_TIMEOUT_HOURS" stage4_status node /tmp/generate-repo-docs.cjs + + # 4. Count results + REPO_DOCS=0 + for f in README.md CONTRIBUTING.md LICENSE.md SECURITY.md; do + if [ -f "$f" ]; then + SIZE=$(wc -c < "$f" | tr -d ' ') + echo "$f: $SIZE bytes" + REPO_DOCS=$((REPO_DOCS + 1)) + fi + done + + set_output "stage4_files" "$REPO_DOCS" + if [ "$REPO_DOCS" -gt 0 ]; then + echo "Repository docs present: $REPO_DOCS" + else + echo "::warning title=Stage 4 produced nothing::No repository docs (README.md, CONTRIBUTING.md, LICENSE.md, SECURITY.md) are present." + fi + + # ========================================================================= + # COMMIT STAGE 4 RESULTS (progressive pull request) + # ========================================================================= + # `!= ''` โ€” the run_stage commit rule, stated once at Stage 1. + - name: Commit and push stage 4 results + if: always() && steps.stage4.outputs.stage4_status != '' + id: commit-stage4 + env: + BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }} + STAGE4_FILES: ${{ steps.stage4.outputs.stage4_files }} + run: | + source /tmp/workflow-helpers.sh + + # .gitignore in each stage 4 managed directory + for managed_dir in docs/api docs/deployment docs/operations docs/cli; do + if [ -d "$managed_dir" ]; then + { + echo "# VoltAgent temp files" + echo "temp/" + echo "" + echo "# JSON intermediate files (except schema/config)" + echo "*.json" + echo "!*-schema.json" + echo "!*-config.json" + } > "$managed_dir/.gitignore" + echo "Wrote $managed_dir/.gitignore" + fi + done + + # Stage repository documentation files + # AGENTS.md (and CLAUDE.md's @AGENTS.md import) are here as the backstop for a run whose Stage 2 commit did not happen. + for f in README.md CONTRIBUTING.md LICENSE.md SECURITY.md AGENTS.md CLAUDE.md; do + if [ -f "$f" ]; then + git add -f "$f" + fi + done + + # Stage Stage 4 managed directories + for managed_dir in docs/api docs/deployment docs/operations docs/cli; do + if [ -d "$managed_dir" ]; then + git add -f "$managed_dir/" 2>/dev/null || true + fi + done + + # Stage docs/README.md if exists + if [ -f "docs/README.md" ]; then + git add -f "docs/README.md" + fi + + # Check if there are changes + STAGED_COUNT=$(git diff --cached --name-only | wc -l) + + if [ "$STAGED_COUNT" -gt 0 ]; then + # Commit and push + git commit -m "docs: Stage 4 - Repository documentation ($STAGE4_FILES files) [skip ci]" + git push origin "$BRANCH_NAME" + + echo "Committed and pushed $STAGED_COUNT stage 4 files" + set_output "committed" "true" + else + echo "No stage 4 changes to commit" + set_output "committed" "false" + fi + + - name: Update pull request with stage 4 progress + if: always() && steps.commit-stage4.outputs.committed == 'true' && steps.create-initial-pr.outputs.pull-request-number + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }} + STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }} + STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }} + STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || 0 }} + STAGE4_FILES: ${{ steps.stage4.outputs.stage4_files || 0 }} + RUN_ID_VAR: ${{ env.RUN_ID }} + REPO_NAME: ${{ github.repository }} + run: | + source /tmp/workflow-helpers.sh + + # Update PR body + PR_BODY="## ๐Ÿฆฉ Flamingo Code Documentation: In Progress + + **Run ID:** \`$RUN_ID_VAR\` + **Status:** ๐Ÿ”„ Running... + + This PR is being updated as each documentation stage completes. + + ### Progress + - โœ… Stage 1: Inline Documentation - Completed ($STAGE1_FILES files) + - โœ… Stage 2: Architecture Analysis - Completed ($STAGE2_FILES files) + - โœ… Stage 3: Tutorial Generation - Completed ($STAGE3_FILES files) + - โœ… Stage 4: Repository Documentation - Completed ($STAGE4_FILES files) + + --- + ๐Ÿฆฉ Generated by [Flamingo Code Documentation](https://flamingo.run)" + + gh pr edit "$PR_NUMBER" --body "$PR_BODY" + + echo "Pull request #$PR_NUMBER updated with stage 4 progress" + + # ========================================================================= + # VALIDATE GENERATED MARKDOWN + # Warn-only validation (never blocks the pull request) + # ========================================================================= + - name: Validate generated Markdown + if: always() + continue-on-error: true # NEVER block PR - validation is warn-only + env: + DOCS_OUTPUT_DIR: ${{ env.DOCS_OUTPUT_PATH }} + # OSS Tenant Structure paths for validation + REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }} + DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }} + GETTING_STARTED_OUTPUT_PATH: ${{ env.GETTING_STARTED_OUTPUT_PATH }} + DEVELOPMENT_OUTPUT_PATH: ${{ env.DEVELOPMENT_OUTPUT_PATH }} + NODE_PATH: ${{ runner.temp }}/doc-orch-deps/node_modules + run: | + # Warn-only: findings are logged per directory and never fail the run. + echo "::group::Validate $DOCS_OUTPUT_DIR" + node /tmp/validate-markdown.js "$DOCS_OUTPUT_DIR" 2>&1 || true + echo "::endgroup::" + # Validate OSS Tenant Structure outputs + echo "::group::Validate $REFERENCE_OUTPUT_PATH" + node /tmp/validate-markdown.js "$REFERENCE_OUTPUT_PATH" 2>&1 || true + echo "::endgroup::" + echo "::group::Validate $GETTING_STARTED_OUTPUT_PATH" + node /tmp/validate-markdown.js "$GETTING_STARTED_OUTPUT_PATH" 2>&1 || true + echo "::endgroup::" + echo "::group::Validate $DEVELOPMENT_OUTPUT_PATH" + node /tmp/validate-markdown.js "$DEVELOPMENT_OUTPUT_PATH" 2>&1 || true + echo "::endgroup::" + + - name: Report stage 4 progress + if: always() && env.HUB_BASE_URL != '' + continue-on-error: true + env: + WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }} + WORKFLOW_RUN_ID: ${{ github.run_id }} + WORKFLOW_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} + STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }} + STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }} + STAGE2_STATUS: ${{ steps.stage2.outputs.stage2_status || steps.stage2_alt.outputs.stage2_status || 'skipped' }} + STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }} + STAGE3_STATUS: ${{ steps.stage3.outputs.stage3_status || 'skipped' }} + STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || 0 }} + STAGE4_STATUS: ${{ steps.stage4.outputs.stage4_status || 'skipped' }} + STAGE4_FILES: ${{ steps.stage4.outputs.stage4_files || 0 }} + PR_URL: ${{ steps.create-initial-pr.outputs.pull-request-url }} + PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }} + run: | + source /tmp/workflow-helpers.sh + + CALLBACK_URL="${HUB_BASE_URL}/api/code-documentation/webhook" + echo "Reporting stage 4 (repository docs): $STAGE4_STATUS, $STAGE4_FILES files" + report_stage_progress "$CALLBACK_URL" "$WEBHOOK_SECRET" "$RUN_ID" "$REPO_ID" \ + "$WORKFLOW_RUN_ID" "$WORKFLOW_URL" "creating-pr" \ + "$STAGE1_STATUS" "$STAGE1_FILES" "$STAGE2_STATUS" "$STAGE2_FILES" "$STAGE3_STATUS" "$STAGE3_FILES" \ + "$STAGE4_STATUS" "$STAGE4_FILES" "$PR_URL" "$PR_NUMBER" + + # ========================================================================= + # FINALIZE THE PULL REQUEST + # ========================================================================= + - name: Clean up temporary files + run: | + source /tmp/workflow-helpers.sh + + # Remove stats files + cleanup_path ".doc-stage1-stats.json" + cleanup_path ".doc-stage3-stats.json" + + # Remove run status file (created for the initial PR). + # Legacy name kept so branches started before the rename still clean up. + cleanup_path ".flamingo-ai-technical-writer-status.md" + cleanup_path ".doc-pipeline-status.md" + + # Note: .code-documentation-source-files.txt is now in /tmp/ (auto-cleanup) + + # Remove npm artifacts (installed for scripts) + cleanup_path "node_modules" + cleanup_path "package.json" + cleanup_path "package-lock.json" + + # NOTE: /tmp/workflow-helpers.sh is removed by "Report run result" + + - name: Stage remaining docs and detect changes + id: stage-docs + run: | + source /tmp/workflow-helpers.sh + echo "Docs root: $DOCS_OUTPUT_PATH" + echo "Stage 2: $REFERENCE_OUTPUT_PATH (reference), $DIAGRAMS_OUTPUT_PATH (diagrams)" + echo "Stage 3: $GETTING_STARTED_OUTPUT_PATH (getting started), $DEVELOPMENT_OUTPUT_PATH (development)" + + # Count untracked/modified files before staging + BEFORE_COUNT=$(git status --porcelain | wc -l) + echo "Changed files in the working tree: $BEFORE_COUNT" + + # Stage ALL .md and .mmd files anywhere in the repo (for inline docs generated next to source files) + # This catches Stage 1 inline docs (hidden: .FileName.md), Stage 2 reference/diagrams, Stage 3 tutorials, and Stage 4 repo docs + # Find all .md and .mmd files recursively, including hidden files (.*.md) + # Includes README.md, CONTRIBUTING.md, LICENSE.md, SECURITY.md from Stage 4 + # Includes .mmd Mermaid diagram files from Stage 2 (CodeWiki/Claude architecture) + find . \( -name "*.md" -o -name ".*.md" -o -name "*.mmd" \) -type f \ + -not -path "./node_modules/*" \ + -not -path "./.git/*" \ + -not -name "CHANGELOG.md" \ + -exec git add -f {} \; 2>/dev/null || true + + echo "::group::Markdown and Mermaid files in the checkout (first 100)" + find . \( -name "*.md" -o -name ".*.md" -o -name "*.mmd" \) -type f \ + -not -path "./node_modules/*" \ + -not -path "./.git/*" \ + -not -name "CHANGELOG.md" | head -100 + echo "::endgroup::" + + # Count staged files + STAGED_COUNT=$(git diff --cached --name-only | wc -l) + echo "Staged files: $STAGED_COUNT" + set_output "staged_count" "$STAGED_COUNT" + + echo "::group::Staged files (first 50)" + git diff --cached --name-only | head -50 + echo "::endgroup::" + + # "Changes" means the BRANCH differs from the documented source head, not + # that this final sweep found something left to stage: every stage commits + # its own output as it goes, so a run whose stages all committed (inline + # docs, README) left nothing here and was reported as no_changes while its + # pull request held twenty files (openframe-saas-mobile run 35302244804). + # The status file is the run's own bookkeeping, never a documentation change. + if [ "$STAGED_COUNT" -eq "0" ] && git diff --quiet "$SOURCE_HEAD_SHA" HEAD -- . ':!.flamingo-ai-technical-writer-status.md'; then + echo "::notice title=No documentation changes::Nothing left to stage, and the branch holds no documentation change against the source head." + set_output "has_changes" "false" + else + set_output "has_changes" "true" + fi + + - name: Mark pull request complete + if: steps.create-initial-pr.outputs.pull-request-number + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }} + STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }} + STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }} + STAGE2_STATUS: ${{ steps.stage2.outputs.stage2_status || steps.stage2_alt.outputs.stage2_status || 'skipped' }} + STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }} + STAGE3_STATUS: ${{ steps.stage3.outputs.stage3_status || 'skipped' }} + STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || 0 }} + STAGE4_STATUS: ${{ steps.stage4.outputs.stage4_status || 'skipped' }} + STAGE4_FILES: ${{ steps.stage4.outputs.stage4_files || 0 }} + RUN_ID_VAR: ${{ env.RUN_ID }} + REPO_NAME: ${{ github.repository }} + run: | + source /tmp/workflow-helpers.sh + + # Remove "in-progress" label + gh pr edit "$PR_NUMBER" --remove-label "in-progress" || true + + # Update title to remove [IN PROGRESS] + gh pr edit "$PR_NUMBER" --title "๐Ÿฆฉ Flamingo Code Documentation" + + # Update body with final results + PR_BODY="## ๐Ÿฆฉ Flamingo Code Documentation: Complete + + **Run ID:** \`$RUN_ID_VAR\` + + ### Stage 1: Inline Documentation + - Status: $STAGE1_STATUS + - Files generated: $STAGE1_FILES + - Generated .md files next to source classes explaining their purpose + + ### Stage 2: Architecture Analysis + - Status: $STAGE2_STATUS + - Files generated: $STAGE2_FILES + - Architecture overview and module documentation + + ### Stage 3: AI Tutorial Generator + - Status: $STAGE3_STATUS + - Files generated: $STAGE3_FILES + - Getting started guides and how-to tutorials + + ### Stage 4: Repository Documentation + - Status: $STAGE4_STATUS + - Files generated: $STAGE4_FILES + - README.md, CONTRIBUTING.md, LICENSE.md, SECURITY.md + + --- + + **Review checklist:** + - [ ] Check generated inline docs for accuracy + - [ ] Review architecture documentation + - [ ] Test code examples in tutorials + - [ ] Review README.md and CONTRIBUTING.md updates + + --- + ๐Ÿฆฉ Generated by [Flamingo Code Documentation](https://flamingo.run)" + + gh pr edit "$PR_NUMBER" --body "$PR_BODY" + + echo "Pull request #$PR_NUMBER marked complete" + + # ========================================================================= + # REPORT RUN RESULT: the terminal callback. Never reports "running". + # ========================================================================= + - name: Report run result + if: always() && env.HUB_BASE_URL != '' + continue-on-error: true # Don't fail the workflow if callback fails + env: + # SECURITY: Pass secret per-step with inline masking + WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }} + WORKFLOW_RUN_ID: ${{ github.run_id }} + WORKFLOW_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} + HAS_CHANGES: ${{ steps.stage-docs.outputs.has_changes }} + JOB_STATUS: ${{ job.status }} + PR_URL: ${{ steps.create-initial-pr.outputs.pull-request-url }} + PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number || 'null' }} + SAFE_RUN_ID: ${{ steps.branch-name-early.outputs.safe_run_id }} + # Single source of truth for the branch name (was rebuilt by hand below, + # which silently drifted from "Create docs branch" on every rename) + BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }} + STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }} + STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }} + # Stage 2: Check both CodeWiki and Claude alternative, mark as failed if step failed + STAGE2_STATUS: ${{ steps.stage2.outcome == 'failure' && 'failed' || steps.stage2_alt.outcome == 'failure' && 'failed' || steps.stage2.outputs.stage2_status || steps.stage2_alt.outputs.stage2_status || 'skipped' }} + STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }} + STAGE3_STATUS: ${{ steps.stage3.outputs.stage3_status || 'skipped' }} + STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || 0 }} + STAGE4_STATUS: ${{ steps.stage4.outputs.stage4_status || 'skipped' }} + STAGE4_FILES: ${{ steps.stage4.outputs.stage4_files || 0 }} + # Track if critical steps failed (continue-on-error: false steps) + STAGE2_OUTCOME: ${{ steps.stage2.outcome || 'skipped' }} + CODEWIKI_INSTALL_OUTCOME: ${{ steps.codewiki_install.outcome || 'skipped' }} + CODEWIKI_CONFIG_OUTCOME: ${{ steps.codewiki_config.outcome || 'skipped' }} + run: | + # Bootstrap-failure fallback (shared failure-net standard with the + # code-review workflow): if workflow-helpers.sh never downloaded, no + # helper exists to report the failure โ€” a minimal guarded curl posts + # it so the hub's run row fails NOW instead of waiting for the reaper. + if [ ! -f /tmp/workflow-helpers.sh ]; then + echo "::error title=Script bootstrap failed::workflow-helpers.sh was never downloaded from the hub; reporting the failure with a minimal callback." + # The bearer goes through a 0600 config file, never argv โ€” see + # curlAuthPreamble in lib/config/workflow-scripts-bootstrap.ts. + CURL_CFG=$(mktemp) && chmod 600 "$CURL_CFG" + trap 'rm -f "$CURL_CFG"' EXIT + printf 'header = "Authorization: Bearer %s"\n' "$WEBHOOK_SECRET" > "$CURL_CFG" + curl -sS --max-time 30 -K "$CURL_CFG" -X POST "${HUB_BASE_URL}/api/code-documentation/webhook" \ + -H "Content-Type: application/json" \ + -d "{\"run_id\":\"$RUN_ID\",\"repo_id\":\"$REPO_ID\",\"status\":\"failure\",\"workflow_run_id\":$WORKFLOW_RUN_ID,\"workflow_url\":\"$WORKFLOW_URL\",\"error\":\"Script bootstrap failed: workflow-helpers.sh never downloaded from the hub.\"}" || true + exit 1 + fi + source /tmp/workflow-helpers.sh + + CALLBACK_URL="${HUB_BASE_URL}/api/code-documentation/webhook" + + echo "::group::Inputs to the final status" + echo "JOB_STATUS=$JOB_STATUS" + echo "HAS_CHANGES=$HAS_CHANGES" + echo "STAGE2_OUTCOME=$STAGE2_OUTCOME" + echo "CODEWIKI_INSTALL_OUTCOME=$CODEWIKI_INSTALL_OUTCOME" + echo "CODEWIKI_CONFIG_OUTCOME=$CODEWIKI_CONFIG_OUTCOME" + echo "::endgroup::" + + # CRITICAL: Determine final status - NEVER return "running" + # Default to failure, only set success if everything checks out + STATUS="failure" + + # Check for cancelled job first + if [ "$JOB_STATUS" = "cancelled" ]; then + STATUS="cancelled" + REASON="the workflow run was cancelled" + # Check if critical stage 2 (CodeWiki) failed - this has continue-on-error: false + elif [ "$STAGE2_OUTCOME" = "failure" ]; then + STATUS="failure" + REASON="the CodeWiki stage failed" + # Check if CodeWiki installation failed + elif [ "$CODEWIKI_INSTALL_OUTCOME" = "failure" ]; then + STATUS="failure" + REASON="the CodeWiki installation failed" + # Check if CodeWiki configuration failed + elif [ "$CODEWIKI_CONFIG_OUTCOME" = "failure" ]; then + STATUS="failure" + REASON="the CodeWiki configuration failed" + # Check overall job status + elif [ "$JOB_STATUS" != "success" ]; then + STATUS="failure" + REASON="the job status is $JOB_STATUS" + # Check if we have any documentation changes + elif [ "$HAS_CHANGES" != "true" ]; then + STATUS="no_changes" + REASON="no documentation changed" + else + STATUS="success" + REASON="documentation generated" + fi + + # SAFETY CHECK: Ensure status is NEVER "running" + if [ "$STATUS" = "running" ] || [ -z "$STATUS" ]; then + echo "::warning::Computed status '$STATUS' is not terminal; reporting failure instead." + STATUS="failure" + REASON="no terminal status could be determined" + fi + + case "$STATUS" in + success) echo "Final status: success ($REASON)" ;; + no_changes) echo "::notice title=Code Documentation: no changes::Final status: no_changes ($REASON)." ;; + cancelled) echo "::warning title=Code Documentation: cancelled::Final status: cancelled ($REASON)." ;; + *) echo "::error title=Code Documentation: $STATUS::Final status: $STATUS ($REASON)." ;; + esac + + # Branch actually created by "Create docs branch". Falls back to the same + # formula only when that step never ran (this step is `if: always()`). + SAFE_BRANCH="${BRANCH_NAME:-docs/flamingo-ai-technical-writer-$SAFE_RUN_ID}" + + # Send final webhook using helper function + report_final_status "$CALLBACK_URL" "$WEBHOOK_SECRET" "$RUN_ID" "$REPO_ID" \ + "$WORKFLOW_RUN_ID" "$WORKFLOW_URL" "$STATUS" "$PR_URL" "$PR_NUMBER" "$SAFE_BRANCH" \ + "$STAGE1_STATUS" "$STAGE1_FILES" "$STAGE2_STATUS" "$STAGE2_FILES" "$STAGE3_STATUS" "$STAGE3_FILES" \ + "$STAGE4_STATUS" "$STAGE4_FILES" + + # Final cleanup: remove workflow helpers file + cleanup_path "/tmp/workflow-helpers.sh" + + # The run's job summary (the Actions run page). Values reach the script + # through env, never inline expressions. + - name: Write job summary + if: always() + env: + STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }} + STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || '0' }} + STAGE2_STATUS: ${{ steps.stage2.outputs.stage2_status || steps.stage2_alt.outputs.stage2_status || 'skipped' }} + STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || '0' }} + STAGE2_ENGINE: ${{ steps.detect_language.outputs.codewiki_supported == 'true' && 'CodeWiki' || steps.detect_language.outputs.codewiki_supported == 'false' && 'Claude' || 'engine not chosen' }} + STAGE3_STATUS: ${{ steps.stage3.outputs.stage3_status || 'skipped' }} + STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || '0' }} + STAGE4_STATUS: ${{ steps.stage4.outputs.stage4_status || 'skipped' }} + STAGE4_FILES: ${{ steps.stage4.outputs.stage4_files || '0' }} + GRAPH_OUTCOME: ${{ steps.graph.outcome || 'skipped' }} + PRIMARY_LANGUAGE: ${{ steps.detect_language.outputs.primary_language }} + PR_URL: ${{ steps.create-initial-pr.outputs.pull-request-url }} + run: | + { + echo "## ๐Ÿฆฉ Flamingo Code Documentation" + echo "" + echo "Run \`$RUN_ID\` ยท source branch \`$SOURCE_BRANCH\` ยท primary language ${PRIMARY_LANGUAGE:-unknown}" + echo "" + echo "| Stage | Status | Files |" + echo "|-------|--------|-------|" + echo "| 0 ยท Code graph | $GRAPH_OUTCOME | โ€“ |" + echo "| 1 ยท Inline docs | $STAGE1_STATUS | $STAGE1_FILES |" + echo "| 2 ยท Reference docs ($STAGE2_ENGINE) | $STAGE2_STATUS | $STAGE2_FILES |" + echo "| 3 ยท Tutorials | $STAGE3_STATUS | $STAGE3_FILES |" + echo "| 4 ยท Repository docs | $STAGE4_STATUS | $STAGE4_FILES |" + echo "" + if [ -n "$PR_URL" ]; then + echo "**Pull request:** $PR_URL" + else + echo "No pull request was opened." + fi + } >> "$GITHUB_STEP_SUMMARY" From 0641e93f1db5c3c09b55d2f79addb3435850c982 Mon Sep 17 00:00:00 2001 From: "flamingo[bot]" <277372822+flamingo[bot]@users.noreply.github.com> Date: Thu, 24 Sep 2026 18:05:14 +0000 Subject: [PATCH 2/2] Remove .github/workflows/doc-orchestrator.yml: the workflow now lives at .github/workflows/code-documentation.yml --- .github/workflows/doc-orchestrator.yml | 3191 ------------------------ 1 file changed, 3191 deletions(-) delete mode 100644 .github/workflows/doc-orchestrator.yml diff --git a/.github/workflows/doc-orchestrator.yml b/.github/workflows/doc-orchestrator.yml deleted file mode 100644 index 8e77473a..00000000 --- a/.github/workflows/doc-orchestrator.yml +++ /dev/null @@ -1,3191 +0,0 @@ -# Flamingo Code Documentation -# ============================================================================= -# Installed into a target repository by the multi-platform hub ("Setup -# Workflow"). The hub dispatches it to generate this repository's -# documentation and open a pull request with the result. -# -# Jobs -# code-graph Deterministic code graph (no model call). Runs on pushes to -# the default branch, on the hub's `flamingo-code-graph` -# re-dispatch, and on a manual run with graph_only=true. -# doc-pipeline The documentation run, dispatched by the hub: -# Stage 0 code graph (same build as the code-graph job) -# Stage 1 inline docs, one hidden .md beside each source file -# Stage 2 reference docs: CodeWiki, or the Claude -# architecture analysis where CodeWiki cannot parse -# the primary language -# Stage 3 tutorials (getting started, development) -# Stage 4 repository docs: README, CONTRIBUTING, docs index -# -# Stages 2 (Claude), 3 and 4 follow the code reviewer's agentic pattern -# (templates/scripts/code-documentation-lib.mjs): the hub serves this -# repository's settings, the script packs the material, ONE hub call writes -# ONE document (forced `emit_document`; the hub's read tools when the -# repository's "Graph lookups" switch is on), a deterministic gate checks every -# repository path it names, and the script, never the model, picks where each -# document is written. -# -# Fleet contracts (renaming any of these needs a fleet-wide re-push): the -# installed path .github/workflows/doc-orchestrator.yml, the -# repository_dispatch type `doc-orchestrator`, the secrets below, and the -# script file names the hub serves. -# -# Repository secrets (Settings > Secrets and variables > Actions) -# DOC_ORCH_WEBHOOK_SECRET Required. Authenticates every hub call. -# ANTHROPIC_API_KEY CodeWiki (stage 2) only; every other stage calls -# Claude through the hub. -# OPENAI_API_KEY CodeWiki (stage 2). -# DOC_ORCH_GITHUB_PAT Optional. Clones private dependency repositories. -# YOUTUBE_API_KEY Optional. YouTube embeds in stages 3 and 4. - -name: ๐Ÿฆฉ Flamingo Code Documentation - -on: - # Push trigger โ€” two things ride it. It registers the workflow with GitHub - # Actions (required for the workflow_dispatch API), and on the repository's - # DEFAULT branch it runs the `code-graph` job below, which re-indexes the - # code graph the hub serves to the code reviewer and to the documentation - # stages. The documentation pipeline itself NEVER runs on push (see its - # `if:`). Documentation and markdown are ignored on purpose: a docs PR - # merging must not rebuild a graph that only source files can change. - push: - paths-ignore: - - 'docs/**' - - '**.md' - - repository_dispatch: - # `doc-orchestrator` runs the documentation pipeline (the type keeps its - # pre-rename spelling: it is a fleet contract). `flamingo-code-graph` - # (CODE_GRAPH_DISPATCH_EVENT_TYPE in lib/config/code-graph-workflow.ts) - # runs ONLY the graph job; the hub's reconcile job sends it when a - # repository's graph is missing or stale. - types: [doc-orchestrator, flamingo-code-graph] - - workflow_dispatch: - # โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ• - # GENERATED FROM SINGLE SOURCE OF TRUTH: lib/config/code-documentation-params.ts - # This section is auto-generated at runtime when creating workflow PRs - # โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ• - inputs: - # Individual parameters - run_id: - description: 'Unique execution ID' - required: true - repo_id: - description: 'Repository ID from database' - required: true - hub_base_url: - description: 'Hub base URL (e.g., https://product-hub.flamingo.so)' - required: true - stages: - description: 'Pipeline stages to execute' - required: true - dependencies: - description: 'Comma-separated dependency repos' - required: false - default: '' - source_branch: - description: 'Branch to analyze code from' - required: true - source_files_limit: - description: 'Max source files to process (0 = unlimited)' - required: true - claude_model: - description: 'Claude model ID for Stage 1/3/4 + Stage 2 Claude-arch fallback (SSOT from hub)' - required: true - codewiki_config: - description: 'Complete CodeWiki configuration (per-phase models, engine, stages, depth)' - required: true - output_paths: - description: 'Output paths configuration' - required: true - timeout: - description: 'Timeout in hours' - required: true - youtube_config: - description: 'YouTube integration configuration (channels only - API key in secrets)' - required: true - readme_config: - description: 'README logo configuration' - required: true - custom_repo_instructions: - description: 'Custom AI instructions' - required: false - default: '' - external_repos: - description: 'External repos JSON' - required: false - default: '[]' - stage_count: - description: 'Total number of pipeline stages' - required: false - default: '4' - graph_only: - description: 'true = run only the code-graph job (no documentation run)' - required: false - default: 'false' - # โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ• - # END GENERATED SECTION - # โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ•โ• - -env: - # SECURITY: Only NON-SENSITIVE variables in job-level env - # Secrets are passed per-step to avoid exposure in job setup logs - # Run configuration (non-sensitive) - RUN_ID: ${{ github.event.client_payload.run_id || github.event.inputs.run_id || github.run_id }} - REPO_ID: ${{ github.event.client_payload.repo_id || github.event.inputs.repo_id || '' }} - # A push event carries no payload, so the graph job falls back to the org - # Actions variable โ€” the same fallback the code-review workflow uses. - HUB_BASE_URL: ${{ github.event.client_payload.hub_base_url || github.event.inputs.hub_base_url || vars.FLAMINGO_HUB_BASE_URL || '' }} - # 'true' = run only the code-graph job (a manual workflow_dispatch; the hub's - # own documentation dispatches always send 'false'). - GRAPH_ONLY: ${{ github.event.client_payload.graph_only || github.event.inputs.graph_only || 'false' }} - # No literal fallback: the stage list lives in CODE_DOCUMENTATION_STAGES on the - # hub and is lifted into the payload per repo. A literal here would silently - # restore all four stages on a payload gap, and "Remove docs this run - # regenerates" would already have wiped the docs tree: the run would delete - # docs and regenerate nothing. - STAGES: ${{ github.event.client_payload.stages || github.event.inputs.stages || '' }} - DEPENDENCIES: ${{ github.event.client_payload.dependencies || github.event.inputs.dependencies || '' }} - STAGE_COUNT: ${{ github.event.client_payload.stage_count || github.event.inputs.stage_count || '4' }} - # Branch to checkout for code analysis (github_branch from repo config) - SOURCE_BRANCH: ${{ github.event.client_payload.source_branch || github.event.inputs.source_branch || 'main' }} - # Debug/testing: limit total source files to analyze (0=unlimited) - # Files beyond this limit are DELETED - all stages then process remaining files - SOURCE_FILES_LIMIT: ${{ github.event.client_payload.source_files_limit || github.event.inputs.source_files_limit || '0' }} - # Claude model SSOT: the hub's CODE_DOCUMENTATION_DEFAULT_MODEL - # (lib/constants/ai-models.ts). The stage 1, 3 and 4 generators and the - # stage 2 Claude analysis all read this variable; no shipped script holds a - # literal model id, and every one throws if CLAUDE_MODEL is empty. Re-run - # "Setup Workflow" after bumping the hub-side constant to propagate it. - # There is deliberately NO companion request-shape variable: every stage - # except CodeWiki calls Claude through the hub (/api/ci/claude), which - # resolves the model's request shape itself. - CLAUDE_MODEL: ${{ github.event.client_payload.claude_model || github.event.inputs.claude_model || '' }} - # ============================================================================= - # JSON-grouped parameters to stay under GitHub Actions 25-parameter limit - # These are parsed early in the workflow to extract individual values - # ============================================================================= - # NOTE: these fall back to EMPTY, not '{}'. An empty-object default made the - # `[ -z ... ]` presence checks below unreachable, so a missing payload silently - # produced `null` for every jq lookup and propagated as `--cluster-model null`. - # The hub always sends a complete, deep-merged blob (buildPayloadFromRepo). - CODEWIKI_CONFIG_JSON: ${{ github.event.client_payload.codewiki_config || github.event.inputs.codewiki_config || '' }} - OUTPUT_PATHS_JSON: ${{ github.event.client_payload.output_paths || github.event.inputs.output_paths || '' }} - # Stage timeouts (in hours) โ€” single source of truth, downstream steps reference - # `env.STAGE_TIMEOUT_HOURS` directly. Stage 4 is the exception (1h vs 24h cap). - STAGE_TIMEOUT_HOURS: ${{ github.event.client_payload.timeout || github.event.inputs.timeout || '24' }} - # Aliases preserved for downstream step env: keys (they reference these names - # by string). All resolve to the same single source. - STAGE1_TIMEOUT_HOURS: ${{ github.event.client_payload.timeout || github.event.inputs.timeout || '24' }} - # Stage 1 incremental push: commit + push the PR branch every N generated inline - # docs. A 24h stage that gets cancelled used to lose ALL of its work because the - # only commit happened after the generator returned. Bound the loss to N files. - STAGE1_PUSH_INTERVAL: ${{ github.event.client_payload.stage1_push_interval || '100' }} - STAGE2_TIMEOUT_HOURS: ${{ github.event.client_payload.timeout || github.event.inputs.timeout || '24' }} - STAGE3_TIMEOUT_HOURS: ${{ github.event.client_payload.timeout || github.event.inputs.timeout || '24' }} - # Stage 4: Repository Documentation - TEMPLATE_REPO: ${{ github.event.client_payload.template_repo || 'flamingo-stack/openframe-oss-tenant' }} - TEMPLATE_BRANCH: ${{ github.event.client_payload.template_branch || 'main' }} - STAGE4_TIMEOUT_HOURS: ${{ github.event.client_payload.stage4_timeout || '1' }} - # YouTube Integration (Stage 3 + Stage 4) - JSONB configuration (API key from secrets) - YOUTUBE_CONFIG_JSON: ${{ github.event.client_payload.youtube_config || github.event.inputs.youtube_config || '' }} - # README Configuration - JSONB configuration for logo branding - README_CONFIG_JSON: ${{ github.event.client_payload.readme_config || github.event.inputs.readme_config || '' }} - # Custom AI Instructions (All Stages) - Repository-specific instructions for AI generation - CUSTOM_REPO_INSTRUCTIONS: ${{ github.event.client_payload.custom_repo_instructions || github.event.inputs.custom_repo_instructions || '' }} - # External Repositories - JSON array of external repo configurations - EXTERNAL_REPOS: ${{ github.event.client_payload.external_repos || github.event.inputs.external_repos || '[]' }} - # ======================================================================== - # Repository Context - CRITICAL for preventing AI URL hallucinations - # These values are passed to ALL AI prompts to ensure correct GitHub URLs - # ======================================================================== - GITHUB_REPOSITORY: ${{ github.repository }} # e.g., "flamingo-stack/openframe-oss-tenant" - GITHUB_REPOSITORY_OWNER: ${{ github.repository_owner }} # e.g., "flamingo-stack" - GITHUB_SERVER_URL: ${{ github.server_url }} # e.g., "https://github.com" - # Analysis Exclusions - Complete array of glob patterns to exclude from repository analysis - # (build artifacts, dependencies, the hub's own checkout, cloned dependency repos) - EXCLUDED_PATHS: '**/node_modules/**,**/.git/**,**/target/**,**/dist/**,**/build/**,**/.next/**,**/out/**,**/coverage/**,**/vendor/**,**/.yalc/**,**/.turbo/**,**/.gradle/**,**/__pycache__/**,**/.terraform/**,**/.venv/**,**/venv/**,**/multi-platform-hub/**,**/deps-*/**' - README_LOGO_ALT: 'OpenFrame Logo' - -jobs: - # =========================================================================== - # CODE GRAPH: deterministic, no model call. Tags every public symbol, - # import and manifest of the checkout (code-graph-build.mjs) and uploads the - # result to the hub, which promotes a default-branch snapshot to `live` and - # serves it to the code reviewer (consumers of a symbol a PR removes) and to - # the documentation stages (the derived ecosystem.md). Runs on every push to - # the default branch, on the hub's `flamingo-code-graph` re-dispatch, and on - # a manual workflow_dispatch with graph_only=true. There is no webhook - # callback: the upload IS the report. The same build also runs as stage 0 of - # a full documentation run (inside doc-pipeline, below). - # =========================================================================== - code-graph: - runs-on: ubuntu-latest - timeout-minutes: 20 - permissions: - contents: read - concurrency: - group: flamingo-code-graph-${{ github.repository }} - cancel-in-progress: true - if: >- - (github.event_name == 'push' && github.ref == format('refs/heads/{0}', github.event.repository.default_branch)) - || github.event.action == 'flamingo-code-graph' - || (github.event_name == 'workflow_dispatch' && github.event.inputs.graph_only == 'true') - - steps: - # Fail LOUD, not silent: a push on a repo whose org never set - # FLAMINGO_HUB_BASE_URL would otherwise curl an empty origin and die with - # an unrelated error. Also normalizes a trailing slash ONCE. - - name: Validate configuration - run: | - if [ -z "$HUB_BASE_URL" ]; then - echo "::error title=Hub URL missing::HUB_BASE_URL is empty. Set the organization Actions variable FLAMINGO_HUB_BASE_URL, or pass hub_base_url in the dispatch payload." - exit 1 - fi - echo "HUB_BASE_URL=${HUB_BASE_URL%/}" >> "$GITHUB_ENV" - echo "Hub: ${HUB_BASE_URL%/}" - - # The shared script bootstrap (byte-mirrored from workflow-scripts-bootstrap.ts, - # asserted by the build gate). BOTH graph scripts are downloaded: the builder - # imports ./code-graph-lib.mjs from its own directory. - - name: Download graph scripts - env: - WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }} - run: | - - # Function to download and verify script - SCRIPT_MANIFEST=/tmp/flamingo-script-manifest.json - # WEBHOOK_SECRET reaches curl through a 0600 config file, never argv โ€” see - # curlAuthPreamble, which always traps the removal. - CURL_CFG=$(mktemp) && chmod 600 "$CURL_CFG" - trap 'rm -f "$CURL_CFG"' EXIT - printf 'header = "Authorization: Bearer %s"\n' "$WEBHOOK_SECRET" > "$CURL_CFG" - # The canonical scripts surface and the pre-rename one. load_script_manifest - # picks whichever this deployment actually serves and pins SCRIPTS_BASE_URL. - CI_SCRIPTS_URL="${HUB_BASE_URL%/}/api/ci/scripts" - LEGACY_SCRIPTS_URL="${HUB_BASE_URL%/}/api/doc-orchestrator/scripts" - - # _try_manifest โ€” 0 loaded, 1 no manifest surface there, 2 fatal. - # The manifest is asked for ONE group: its keys are the files to download. - _try_manifest() { - local base="$1" code - code=$(curl -sS -w '%{http_code}' -o "$SCRIPT_MANIFEST" \ - -K "$CURL_CFG" \ - "$base/manifest.json?group=$SCRIPT_GROUP") || code="000" - - if [ "$code" = "404" ]; then rm -f "$SCRIPT_MANIFEST"; return 1; fi - if [ "$code" != "200" ]; then - echo "โŒ manifest request to $base failed (HTTP $code)" - rm -f "$SCRIPT_MANIFEST" - return 2 - fi - # The digests are the TOP-LEVEL object. successResponse is the standard - # emitter but it does NOT add a wrapper โ€” it is NextResponse.json(data) - # plus the no-store header โ€” so there is no .data to reach through. - # A 200 that is not a manifest is how a hub which does not serve this path - # answers (the proxy rewrites unknown routes and returns HTML), so it - # means "wrong surface", not "corrupt". - if ! jq -e 'type == "object" and length > 0 and (to_entries | all(.value | type == "string"))' "$SCRIPT_MANIFEST" >/dev/null 2>&1; then - rm -f "$SCRIPT_MANIFEST" - return 1 - fi - return 0 - } - - # load_script_manifest - load_script_manifest() { - SCRIPT_GROUP="$1" - # The scripts surface was renamed from /api/doc-orchestrator/scripts to the - # pipeline-neutral /api/ci/scripts (it always served BOTH pipelines). The - # workflow file ships in the repo and the routes ship with the deployment, - # so the two are one version apart in BOTH directions across the rollout. - # Probe the canonical surface, fall back to the legacy one, and let the - # winner decide SCRIPTS_BASE_URL for every download that follows. - # "cmd; rc=$?" dies under the set -euo pipefail these steps run with โ€” - # errexit fires before rc is read and the step ends with NO output. And - # "if ! cmd; then rc=$?" is worse: inside the branch $? is the status of - # the NEGATION (0), so every failure reads as success. "|| rc=$?" is the - # one form that both suppresses errexit and preserves the real code. - local rc=0 - _try_manifest "$CI_SCRIPTS_URL" || rc=$? - if [ "$rc" = "0" ]; then - SCRIPTS_BASE_URL="$CI_SCRIPTS_URL" - echo "โœ… script manifest loaded ($(jq -r 'length' "$SCRIPT_MANIFEST") scripts)" - return 0 - fi - if [ "$rc" = "2" ]; then exit 1; fi - - echo "::warning::this hub does not serve $CI_SCRIPTS_URL โ€” falling back to the legacy $LEGACY_SCRIPTS_URL; it predates the rename" - SCRIPTS_BASE_URL="$LEGACY_SCRIPTS_URL" - - rc=0 - _try_manifest "$LEGACY_SCRIPTS_URL" || rc=$? - if [ "$rc" = "0" ]; then - echo "โœ… script manifest loaded ($(jq -r 'length' "$SCRIPT_MANIFEST") scripts)" - return 0 - fi - if [ "$rc" = "2" ]; then exit 1; fi - - # Neither surface published a manifest: a hub older than the manifest - # itself. The manifest is the file list, so there is nothing to download. - echo "โŒ no script manifest on this hub ($HUB_BASE_URL): it cannot name the $SCRIPT_GROUP scripts. Redeploy the hub." - exit 1 - } - - # download_script_group โ€” the hub names the files, this workflow - # names only the group. Downloads every script of the group, in served order. - download_script_group() { - load_script_manifest "$1" - local name - # The loop runs in THIS shell (no pipe), so a failed download exits the step. - while IFS= read -r name; do - download_and_verify "$name" - done < <(jq -r 'keys_unsorted[]' "$SCRIPT_MANIFEST") - } - - download_and_verify() { - local script_name="$1" - local output_path="/tmp/$script_name" - - local expected_hash - expected_hash=$(jq -r --arg n "$script_name" '.[$n] // empty' "$SCRIPT_MANIFEST") - if [ -z "$expected_hash" ]; then - echo "โŒ $script_name is not in the server's script manifest!" - echo " The hub serves no such script, or it failed to read on the server." - exit 1 - fi - if ! printf '%s' "$expected_hash" | grep -Eq '^[0-9a-f]{64}$'; then - echo "โŒ the manifest entry for $script_name is not a SHA-256 digest โ€” refusing to run it." - exit 1 - fi - - curl -fsSL "$SCRIPTS_BASE_URL/$script_name" \ - -K "$CURL_CFG" \ - -o "$output_path" - - local actual_hash=$(shasum -a 256 "$output_path" | cut -d' ' -f1) - - if [ "$actual_hash" != "$expected_hash" ]; then - echo "โŒ HASH MISMATCH for $script_name!" - echo " Expected: $expected_hash" - echo " Actual: $actual_hash" - echo " The download was corrupted in transit โ€” both values come from the same deployment." - exit 1 - fi - - # Make shell scripts executable - if [[ "$script_name" == *.sh ]]; then - chmod +x "$output_path" - fi - - echo "โœ… $script_name verified (hash: ${actual_hash:0:16}...)" - } - - # Digests AND the file list come from the deployment serving the bytes, - # not from this file: the step names a group (SCRIPT_GROUPS in the hub's - # lib/config/ci-script-catalog.ts) and downloads what the hub lists for it. - download_script_group "code-graph" - - # FULL history, blobless. `collectFileFacts` derives per-file ownership - # (last commit, recent commits, commits in the window) from one - # `git log --no-merges --no-renames` walk, which a depth-1 checkout - # cannot answer โ€” it would report every file as owned by one commit. - # `filter: blob:none` keeps the clone cheap: the walk reads commit - # metadata and name-only paths, never file contents, which is also why - # the walk passes `--no-renames` (rename detection would fetch blobs). - # A shallow checkout still degrades gracefully: ownership is omitted and - # `coverage.ownership` is false rather than the job failing. - - name: Check out repository - uses: actions/checkout@fbc6f3992d24b796d5a048ff273f7fcc4a7b6c09 # v5.1.0 - with: - fetch-depth: 0 - filter: blob:none - persist-credentials: false - - - name: Set up Node.js - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0 - with: - node-version: '22' - - # The command is CODE_GRAPH_INSTALL_COMMAND (lib/config/code-graph-workflow.ts), - # the one spelling both workflows use: pinned wasm tree-sitter + grammars + - # yaml into an isolated tree under RUNNER_TEMP, exported as CODE_GRAPH_DEPS_DIR. - - name: Install graph dependencies - run: mkdir -p "$RUNNER_TEMP/code-graph-deps" && cd "$RUNNER_TEMP/code-graph-deps" && printf '%s' '{"name":"code-graph-deps","version":"1.0.0","private":true,"dependencies":{"web-tree-sitter":"0.27.0","@vscode/tree-sitter-wasm":"0.3.1","yaml":"2.9.1"}}' > package.json && printf '%s' '{"name":"code-graph-deps","version":"1.0.0","lockfileVersion":3,"requires":true,"packages":{"":{"name":"code-graph-deps","version":"1.0.0","dependencies":{"web-tree-sitter":"0.27.0","@vscode/tree-sitter-wasm":"0.3.1","yaml":"2.9.1"}},"node_modules/web-tree-sitter":{"version":"0.27.0","resolved":"https://registry.npmjs.org/web-tree-sitter/-/web-tree-sitter-0.27.0.tgz","integrity":"sha512-XK08gj6RwTMQatAG7uVRP8MunqotL/XC19vHgkSPKmELgbGPBj4ECvB8haHOUnyj6ls2B8t42UTro14zxGgAHg=="},"node_modules/@vscode/tree-sitter-wasm":{"version":"0.3.1","resolved":"https://registry.npmjs.org/@vscode/tree-sitter-wasm/-/tree-sitter-wasm-0.3.1.tgz","integrity":"sha512-RJFoomET6FajjG511fmQxeBQfU6M24a0aFZPqpid+ttIxanWf1VGytBG0UmsGjt07qmIPJS8U31D+aecuCucsQ=="},"node_modules/yaml":{"version":"2.9.1","resolved":"https://registry.npmjs.org/yaml/-/yaml-2.9.1.tgz","integrity":"sha512-3NxN8+78OdzbT7C/WjGsyfPAtJaN3FNDsWxv7Y7mcDsT/oOmgW8BpyQQFFBnvZE3j9Y2Sdz1ULFLezL7Eb2yFw=="}}}' > package-lock.json && npm ci --ignore-scripts --no-audit --no-fund && echo "CODE_GRAPH_DEPS_DIR=$RUNNER_TEMP/code-graph-deps" >> "$GITHUB_ENV" || { echo "::warning::graph dependencies failed their lockfile-enforced install; continuing without them"; rm -rf "$RUNNER_TEMP/code-graph-deps"; exit 1; } - - - name: Build and upload the code graph - id: graph - env: - WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }} - CODE_GRAPH_DEPS_DIR: ${{ env.CODE_GRAPH_DEPS_DIR }} - GITHUB_REPOSITORY: ${{ github.repository }} - run: node /tmp/code-graph-build.mjs - - # =========================================================================== - # DOCUMENTATION PIPELINE: stages 0 to 4, one pull request per run. Every - # stage commits and pushes its own output as it finishes, so a timeout or a - # cancellation loses at most the stage in flight. - # =========================================================================== - doc-pipeline: - runs-on: ubuntu-latest - timeout-minutes: 720 # 12 hours for large repositories with many files - # contents: push the docs branch. pull-requests: open and update the pull - # request. issues: `gh label create` for the documentation / automated / - # in-progress labels; without it a repository that lacks them answers 403 - # on the label and then 422 on `gh pr create --label`. - permissions: - contents: write - pull-requests: write - issues: write - # Never on push (a push registers the workflow and runs the code-graph job - # only), never on the graph-only re-dispatch, never on a graph-only manual - # run. This is what lets workflow_dispatch API calls work on feature branches. - if: github.event_name != 'push' && github.event.action != 'flamingo-code-graph' && github.event.inputs.graph_only != 'true' - - steps: - # ========================================================================= - # REPORT CAPABILITY FIRST (shared failure-net standard with the code-review - # workflow): workflow-helpers.sh โ€” which carries send_webhook and the - # report/stage-callback helpers โ€” downloads in its OWN step before anything - # else, so a failure in the main script download below can still be pinged - # and reported home instead of leaving a phantom pending/running row. - # ========================================================================= - - name: Download report helpers - env: - WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }} - run: | - - # Function to download and verify script - SCRIPT_MANIFEST=/tmp/flamingo-script-manifest.json - # WEBHOOK_SECRET reaches curl through a 0600 config file, never argv โ€” see - # curlAuthPreamble, which always traps the removal. - CURL_CFG=$(mktemp) && chmod 600 "$CURL_CFG" - trap 'rm -f "$CURL_CFG"' EXIT - printf 'header = "Authorization: Bearer %s"\n' "$WEBHOOK_SECRET" > "$CURL_CFG" - # The canonical scripts surface and the pre-rename one. load_script_manifest - # picks whichever this deployment actually serves and pins SCRIPTS_BASE_URL. - CI_SCRIPTS_URL="${HUB_BASE_URL%/}/api/ci/scripts" - LEGACY_SCRIPTS_URL="${HUB_BASE_URL%/}/api/doc-orchestrator/scripts" - - # _try_manifest โ€” 0 loaded, 1 no manifest surface there, 2 fatal. - # The manifest is asked for ONE group: its keys are the files to download. - _try_manifest() { - local base="$1" code - code=$(curl -sS -w '%{http_code}' -o "$SCRIPT_MANIFEST" \ - -K "$CURL_CFG" \ - "$base/manifest.json?group=$SCRIPT_GROUP") || code="000" - - if [ "$code" = "404" ]; then rm -f "$SCRIPT_MANIFEST"; return 1; fi - if [ "$code" != "200" ]; then - echo "โŒ manifest request to $base failed (HTTP $code)" - rm -f "$SCRIPT_MANIFEST" - return 2 - fi - # The digests are the TOP-LEVEL object. successResponse is the standard - # emitter but it does NOT add a wrapper โ€” it is NextResponse.json(data) - # plus the no-store header โ€” so there is no .data to reach through. - # A 200 that is not a manifest is how a hub which does not serve this path - # answers (the proxy rewrites unknown routes and returns HTML), so it - # means "wrong surface", not "corrupt". - if ! jq -e 'type == "object" and length > 0 and (to_entries | all(.value | type == "string"))' "$SCRIPT_MANIFEST" >/dev/null 2>&1; then - rm -f "$SCRIPT_MANIFEST" - return 1 - fi - return 0 - } - - # load_script_manifest - load_script_manifest() { - SCRIPT_GROUP="$1" - # The scripts surface was renamed from /api/doc-orchestrator/scripts to the - # pipeline-neutral /api/ci/scripts (it always served BOTH pipelines). The - # workflow file ships in the repo and the routes ship with the deployment, - # so the two are one version apart in BOTH directions across the rollout. - # Probe the canonical surface, fall back to the legacy one, and let the - # winner decide SCRIPTS_BASE_URL for every download that follows. - # "cmd; rc=$?" dies under the set -euo pipefail these steps run with โ€” - # errexit fires before rc is read and the step ends with NO output. And - # "if ! cmd; then rc=$?" is worse: inside the branch $? is the status of - # the NEGATION (0), so every failure reads as success. "|| rc=$?" is the - # one form that both suppresses errexit and preserves the real code. - local rc=0 - _try_manifest "$CI_SCRIPTS_URL" || rc=$? - if [ "$rc" = "0" ]; then - SCRIPTS_BASE_URL="$CI_SCRIPTS_URL" - echo "โœ… script manifest loaded ($(jq -r 'length' "$SCRIPT_MANIFEST") scripts)" - return 0 - fi - if [ "$rc" = "2" ]; then exit 1; fi - - echo "::warning::this hub does not serve $CI_SCRIPTS_URL โ€” falling back to the legacy $LEGACY_SCRIPTS_URL; it predates the rename" - SCRIPTS_BASE_URL="$LEGACY_SCRIPTS_URL" - - rc=0 - _try_manifest "$LEGACY_SCRIPTS_URL" || rc=$? - if [ "$rc" = "0" ]; then - echo "โœ… script manifest loaded ($(jq -r 'length' "$SCRIPT_MANIFEST") scripts)" - return 0 - fi - if [ "$rc" = "2" ]; then exit 1; fi - - # Neither surface published a manifest: a hub older than the manifest - # itself. The manifest is the file list, so there is nothing to download. - echo "โŒ no script manifest on this hub ($HUB_BASE_URL): it cannot name the $SCRIPT_GROUP scripts. Redeploy the hub." - exit 1 - } - - # download_script_group โ€” the hub names the files, this workflow - # names only the group. Downloads every script of the group, in served order. - download_script_group() { - load_script_manifest "$1" - local name - # The loop runs in THIS shell (no pipe), so a failed download exits the step. - while IFS= read -r name; do - download_and_verify "$name" - done < <(jq -r 'keys_unsorted[]' "$SCRIPT_MANIFEST") - } - - download_and_verify() { - local script_name="$1" - local output_path="/tmp/$script_name" - - local expected_hash - expected_hash=$(jq -r --arg n "$script_name" '.[$n] // empty' "$SCRIPT_MANIFEST") - if [ -z "$expected_hash" ]; then - echo "โŒ $script_name is not in the server's script manifest!" - echo " The hub serves no such script, or it failed to read on the server." - exit 1 - fi - if ! printf '%s' "$expected_hash" | grep -Eq '^[0-9a-f]{64}$'; then - echo "โŒ the manifest entry for $script_name is not a SHA-256 digest โ€” refusing to run it." - exit 1 - fi - - curl -fsSL "$SCRIPTS_BASE_URL/$script_name" \ - -K "$CURL_CFG" \ - -o "$output_path" - - local actual_hash=$(shasum -a 256 "$output_path" | cut -d' ' -f1) - - if [ "$actual_hash" != "$expected_hash" ]; then - echo "โŒ HASH MISMATCH for $script_name!" - echo " Expected: $expected_hash" - echo " Actual: $actual_hash" - echo " The download was corrupted in transit โ€” both values come from the same deployment." - exit 1 - fi - - # Make shell scripts executable - if [[ "$script_name" == *.sh ]]; then - chmod +x "$output_path" - fi - - echo "โœ… $script_name verified (hash: ${actual_hash:0:16}...)" - } - - # Digests AND the file list come from the deployment serving the bytes, not from this file. - download_script_group "doc-helpers" - - # ========================================================================= - # REPORT RUN STARTED: the early "the workflow actually started" ping. - # Deliberately BEFORE the main script download: it stamps workflow_run_id + - # status 'running' on the hub's run row, which is what lets the hub's tiered - # reaper tell "dispatch accepted but nothing ran" (never pinged, failed - # fast) from "started and then crashed" (pinged, longer deadline). - # ========================================================================= - - name: Report run started - if: env.HUB_BASE_URL != '' - continue-on-error: true - env: - WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }} - run: | - source /tmp/workflow-helpers.sh - - CALLBACK_URL="${HUB_BASE_URL}/api/code-documentation/webhook" - echo "Reporting run start to $CALLBACK_URL" - - PAYLOAD="{ - \"run_id\": \"$RUN_ID\", - \"repo_id\": \"$REPO_ID\", - \"status\": \"running\", - \"workflow_run_id\": ${{ github.run_id }}, - \"workflow_url\": \"${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}\", - \"current_stage\": \"inline-docs\" - }" - - HTTP_CODE=$(send_webhook "$CALLBACK_URL" "$WEBHOOK_SECRET" "$PAYLOAD" "/tmp/webhook_start_response.txt") || HTTP_CODE="failed" - - if [ "$HTTP_CODE" = "200" ] || [ "$HTTP_CODE" = "201" ]; then - echo "Hub acknowledged the run start (HTTP $HTTP_CODE)" - else - echo "::warning title=Run start not reported::The hub answered HTTP $HTTP_CODE to the start callback. The run continues; the hub learns its status from the stage callbacks." - fi - - # ========================================================================= - # DOWNLOAD PIPELINE SCRIPTS - # Every script a documentation run uses, from the hub's authenticated - # /api/ci/scripts surface (workflow-helpers.sh arrived in "Download report - # helpers"), plus the source vocabulary and the Markdown guidelines. - # ========================================================================= - - name: Download pipeline scripts - id: helpers - env: - WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }} - # No HASH_* pins: the digests come from manifest.json on the same - # endpoint that serves the scripts, so a hash in this file can never - # be a different ref's than the bytes it checks. - run: | - echo "::group::Download the doc-pipeline script group" - - # Function to download and verify script - SCRIPT_MANIFEST=/tmp/flamingo-script-manifest.json - # WEBHOOK_SECRET reaches curl through a 0600 config file, never argv โ€” see - # curlAuthPreamble, which always traps the removal. - CURL_CFG=$(mktemp) && chmod 600 "$CURL_CFG" - trap 'rm -f "$CURL_CFG"' EXIT - printf 'header = "Authorization: Bearer %s"\n' "$WEBHOOK_SECRET" > "$CURL_CFG" - # The canonical scripts surface and the pre-rename one. load_script_manifest - # picks whichever this deployment actually serves and pins SCRIPTS_BASE_URL. - CI_SCRIPTS_URL="${HUB_BASE_URL%/}/api/ci/scripts" - LEGACY_SCRIPTS_URL="${HUB_BASE_URL%/}/api/doc-orchestrator/scripts" - - # _try_manifest โ€” 0 loaded, 1 no manifest surface there, 2 fatal. - # The manifest is asked for ONE group: its keys are the files to download. - _try_manifest() { - local base="$1" code - code=$(curl -sS -w '%{http_code}' -o "$SCRIPT_MANIFEST" \ - -K "$CURL_CFG" \ - "$base/manifest.json?group=$SCRIPT_GROUP") || code="000" - - if [ "$code" = "404" ]; then rm -f "$SCRIPT_MANIFEST"; return 1; fi - if [ "$code" != "200" ]; then - echo "โŒ manifest request to $base failed (HTTP $code)" - rm -f "$SCRIPT_MANIFEST" - return 2 - fi - # The digests are the TOP-LEVEL object. successResponse is the standard - # emitter but it does NOT add a wrapper โ€” it is NextResponse.json(data) - # plus the no-store header โ€” so there is no .data to reach through. - # A 200 that is not a manifest is how a hub which does not serve this path - # answers (the proxy rewrites unknown routes and returns HTML), so it - # means "wrong surface", not "corrupt". - if ! jq -e 'type == "object" and length > 0 and (to_entries | all(.value | type == "string"))' "$SCRIPT_MANIFEST" >/dev/null 2>&1; then - rm -f "$SCRIPT_MANIFEST" - return 1 - fi - return 0 - } - - # load_script_manifest - load_script_manifest() { - SCRIPT_GROUP="$1" - # The scripts surface was renamed from /api/doc-orchestrator/scripts to the - # pipeline-neutral /api/ci/scripts (it always served BOTH pipelines). The - # workflow file ships in the repo and the routes ship with the deployment, - # so the two are one version apart in BOTH directions across the rollout. - # Probe the canonical surface, fall back to the legacy one, and let the - # winner decide SCRIPTS_BASE_URL for every download that follows. - # "cmd; rc=$?" dies under the set -euo pipefail these steps run with โ€” - # errexit fires before rc is read and the step ends with NO output. And - # "if ! cmd; then rc=$?" is worse: inside the branch $? is the status of - # the NEGATION (0), so every failure reads as success. "|| rc=$?" is the - # one form that both suppresses errexit and preserves the real code. - local rc=0 - _try_manifest "$CI_SCRIPTS_URL" || rc=$? - if [ "$rc" = "0" ]; then - SCRIPTS_BASE_URL="$CI_SCRIPTS_URL" - echo "โœ… script manifest loaded ($(jq -r 'length' "$SCRIPT_MANIFEST") scripts)" - return 0 - fi - if [ "$rc" = "2" ]; then exit 1; fi - - echo "::warning::this hub does not serve $CI_SCRIPTS_URL โ€” falling back to the legacy $LEGACY_SCRIPTS_URL; it predates the rename" - SCRIPTS_BASE_URL="$LEGACY_SCRIPTS_URL" - - rc=0 - _try_manifest "$LEGACY_SCRIPTS_URL" || rc=$? - if [ "$rc" = "0" ]; then - echo "โœ… script manifest loaded ($(jq -r 'length' "$SCRIPT_MANIFEST") scripts)" - return 0 - fi - if [ "$rc" = "2" ]; then exit 1; fi - - # Neither surface published a manifest: a hub older than the manifest - # itself. The manifest is the file list, so there is nothing to download. - echo "โŒ no script manifest on this hub ($HUB_BASE_URL): it cannot name the $SCRIPT_GROUP scripts. Redeploy the hub." - exit 1 - } - - # download_script_group โ€” the hub names the files, this workflow - # names only the group. Downloads every script of the group, in served order. - download_script_group() { - load_script_manifest "$1" - local name - # The loop runs in THIS shell (no pipe), so a failed download exits the step. - while IFS= read -r name; do - download_and_verify "$name" - done < <(jq -r 'keys_unsorted[]' "$SCRIPT_MANIFEST") - } - - download_and_verify() { - local script_name="$1" - local output_path="/tmp/$script_name" - - local expected_hash - expected_hash=$(jq -r --arg n "$script_name" '.[$n] // empty' "$SCRIPT_MANIFEST") - if [ -z "$expected_hash" ]; then - echo "โŒ $script_name is not in the server's script manifest!" - echo " The hub serves no such script, or it failed to read on the server." - exit 1 - fi - if ! printf '%s' "$expected_hash" | grep -Eq '^[0-9a-f]{64}$'; then - echo "โŒ the manifest entry for $script_name is not a SHA-256 digest โ€” refusing to run it." - exit 1 - fi - - curl -fsSL "$SCRIPTS_BASE_URL/$script_name" \ - -K "$CURL_CFG" \ - -o "$output_path" - - local actual_hash=$(shasum -a 256 "$output_path" | cut -d' ' -f1) - - if [ "$actual_hash" != "$expected_hash" ]; then - echo "โŒ HASH MISMATCH for $script_name!" - echo " Expected: $expected_hash" - echo " Actual: $actual_hash" - echo " The download was corrupted in transit โ€” both values come from the same deployment." - exit 1 - fi - - # Make shell scripts executable - if [[ "$script_name" == *.sh ]]; then - chmod +x "$output_path" - fi - - echo "โœ… $script_name verified (hash: ${actual_hash:0:16}...)" - } - - # Every script a documentation run uses, stage 0 (the code graph) included. - # Digests AND the file list come from the deployment serving the bytes, - # not from this file: the hub lists the group in download order (a helper - # a generator require()s at load comes before it). - download_script_group "doc-pipeline" - echo "::endgroup::" - - # ONE vocabulary read for the whole run: every later step (language - # detection, source discovery, the generators, the graph build) reads this - # file, so none of them needs the secret for it. - node /tmp/ci-source.mjs vocabulary /tmp/ci-vocabulary.json || { echo "::error title=Source vocabulary unavailable::Could not read the source vocabulary from the hub."; exit 1; } - echo "CODE_GRAPH_VOCABULARY_FILE=/tmp/ci-vocabulary.json" >> $GITHUB_ENV - - echo "Pipeline scripts and source vocabulary downloaded and verified" - - # Export paths for all stages (use os.tmpdir() compatible paths) - echo "VALIDATION_RULES_PATH=/tmp/markdown-validation-rules.md" >> $GITHUB_ENV - echo "GUIDELINES_PATH=/tmp/flamingo-markdown-guidelines.md" >> $GITHUB_ENV - echo "STAGE3_FILES_TRACKER=/tmp/stage3-files.txt" >> $GITHUB_ENV - echo "STAGE3_STATS_FILE=/tmp/.doc-stage3-stats.json" >> $GITHUB_ENV - echo "STAGE4_FILES_TRACKER=/tmp/stage4-files.txt" >> $GITHUB_ENV - - # Flamingo Markdown guidelines (their own endpoint). REQUIRED: the - # Markdown validation and the CodeWiki prompts both read them. - GUIDELINES_URL="${HUB_BASE_URL}/api/code-documentation/guidelines" - # Same 0600 config file the download block above set up. - HTTP_CODE=$(curl -fsSL -w "%{http_code}" \ - "$GUIDELINES_URL" \ - -K "$CURL_CFG" \ - -o "/tmp/flamingo-markdown-guidelines.md" 2>/dev/null) || HTTP_CODE="failed" - - if [ "$HTTP_CODE" = "200" ]; then - GUIDELINES_SIZE=$(wc -c < /tmp/flamingo-markdown-guidelines.md | tr -d ' ') - if [ "$GUIDELINES_SIZE" -lt 100 ]; then - echo "::error title=Markdown guidelines invalid::The guidelines file is $GUIDELINES_SIZE bytes, which is an error response, not guidelines. Its body follows." - cat /tmp/flamingo-markdown-guidelines.md - exit 1 - fi - echo "Markdown guidelines downloaded ($GUIDELINES_SIZE bytes)" - else - echo "::error title=Markdown guidelines unavailable::GET $GUIDELINES_URL answered HTTP $HTTP_CODE. The Markdown validation and the CodeWiki prompts require them; check that the hub serves the guidelines endpoint." - rm -f /tmp/flamingo-markdown-guidelines.md - exit 1 - fi - - # (The run-started report sits ABOVE the main script download; see the - # report-capability step ordering at the top of the job.) - - - name: Check out repository - # v5 = the Node 24 drop-in (v4 targets EOL Node 20 and warns on every run). - uses: actions/checkout@fbc6f3992d24b796d5a048ff273f7fcc4a7b6c09 # v5.1.0 - with: - fetch-depth: 0 - token: ${{ secrets.GITHUB_TOKEN }} - ref: ${{ env.SOURCE_BRANCH }} - - # The SOURCE head, before the PR branch and the docs-removal commit move - # HEAD: the stage-0 graph build tags this commit (the code being - # documented), never the docs branch it is sitting on. - - name: Record source head - run: echo "SOURCE_HEAD_SHA=$(git rev-parse HEAD)" >> "$GITHUB_ENV" - - # ========================================================================= - # DEPENDENCY REPOSITORIES (when configured): cloned beside the checkout as - # context for every stage. - # ========================================================================= - - name: Clone dependency repositories - if: env.DEPENDENCIES != '' - env: - GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} - # SECURITY: Pass secret per-step with inline masking - GITHUB_PAT: ${{ secrets.DOC_ORCH_GITHUB_PAT }} - REPO_IS_PRIVATE: ${{ github.event.repository.private }} - run: | - source /tmp/workflow-helpers.sh - - echo "Cloning dependency repositories: $DEPENDENCIES" - ensure_directory "../deps" - - # Determine which token to use. A PUBLIC repository's run NEVER uses - # the PAT: the hub already withholds private dependencies from it, - # and the PAT is what would let a private clone succeed anyway if a - # private slug ever reached this list (defence in depth). - if [ "$REPO_IS_PRIVATE" != "true" ]; then - echo "Credential: GITHUB_TOKEN (public repository, the PAT is never used)" - CLONE_TOKEN="$GH_TOKEN" - elif [ -n "$GITHUB_PAT" ]; then - echo "Credential: DOC_ORCH_GITHUB_PAT (cross-repository access)" - CLONE_TOKEN="$GITHUB_PAT" - else - echo "Credential: GITHUB_TOKEN (no DOC_ORCH_GITHUB_PAT set; private dependencies may fail to clone)" - CLONE_TOKEN="$GH_TOKEN" - fi - - # Configure git to use token for private repos - git config --global url."https://x-access-token:${CLONE_TOKEN}@github.com/".insteadOf "https://github.com/" - - IFS=',' read -ra DEPS <<< "$DEPENDENCIES" - for dep in "${DEPS[@]}"; do - repo_name=$(basename $dep) - echo "::group::Clone $dep into ../deps/$repo_name" - if git clone --depth 1 "https://github.com/$dep.git" "../deps/$repo_name" 2>&1; then - file_count=$(node /tmp/ci-source.mjs count "../deps/$repo_name" 2>/dev/null || echo "?") - echo "Cloned $dep ($file_count source files)" - else - echo "::warning title=Dependency not cloned::Could not clone $dep. A private dependency needs DOC_ORCH_GITHUB_PAT with the repo scope." - fi - echo "::endgroup::" - done - - echo "::group::Dependency directories" - ls -la ../deps/ 2>/dev/null || echo "No dependencies cloned" - echo "::endgroup::" - echo "Dependency source files available as context: $(node /tmp/ci-source.mjs count ../deps 2>/dev/null || echo '?')" - - # ========================================================================= - # PRIMARY LANGUAGE (before every stage, so all of them agree): picks the - # stage 2 engine and filters stages 1 to 3. - # ========================================================================= - - name: Detect primary language - id: detect_language - run: | - source /tmp/workflow-helpers.sh - - # This step is the FIRST reader of CODEWIKI_CONFIG_JSON โ€” it runs before - # "Validate and parse run configuration", so the emptiness guard lives here, - # ahead of the first jq, rather than in the later validation step. - if [ -z "$CODEWIKI_CONFIG_JSON" ]; then - echo "::error title=Missing configuration::CODEWIKI_CONFIG_JSON is empty; language detection needs it." - exit 1 - fi - - echo "Detecting the primary language across the repository (.) and its dependencies (../deps/)" - - # ONE detection, from the vocabulary the hub serves (ci-source.mjs): which - # languages are source, their extensions, and which of them CodeWiki can - # parse are rows in the hub's language table. This step used to carry nine - # hand-typed `find` counts, a positional helper and a six-way threshold - # test, each with its own idea of the extensions and the exclusions. - DETECTION=$(node /tmp/ci-source.mjs detect) || { echo "::error title=Language detection failed::ci-source.mjs detect exited non-zero."; exit 1; } - PRIMARY_LANG=$(echo "$DETECTION" | jq -r '.primary') - MAX_COUNT=$(echo "$DETECTION" | jq -r '.max') - CODEWIKI_SUPPORTED=$(echo "$DETECTION" | jq -r '.codewiki_supported') - - echo "::group::Source files by language (tests and never-source directories excluded)" - echo "$DETECTION" | jq -r '.counts | to_entries[] | select(.value > 0) | " \(.key): \(.value)"' - echo "::endgroup::" - echo "Primary language: $PRIMARY_LANG ($MAX_COUNT files)" - if [ "$CODEWIKI_SUPPORTED" = "true" ]; then - echo "CodeWiki can parse it: yes" - else - echo "CodeWiki can parse it: no (stage 2 uses the Claude architecture analysis)" - fi - - # Per-repo engine override. - # - # The detection above cannot see mixed repos: it counts `.` AND `../deps`, so - # a Rust or Go product with a TypeScript dependency clones its way past the - # >=10 threshold and runs CodeWiki over a codebase whose analyzers do not - # exist โ€” which yields synthetic module_1/module_2/... docs that look like a - # successful run. `engine` pins the choice. - # - # Applied here, before set_output, so all five downstream gates keep reading - # one value and need no change. It cannot live in the `if:` conditions: - # GitHub Actions expressions have no ternary. - CODEWIKI_ENGINE=$(require_json_key "$CODEWIKI_CONFIG_JSON" '.engine' 'codewiki engine') || exit 1 - case "$CODEWIKI_ENGINE" in - claude) - CODEWIKI_SUPPORTED="false" - echo "Stage 2 engine: claude (pinned by the repository configuration)" - ;; - codewiki) - CODEWIKI_SUPPORTED="true" - echo "Stage 2 engine: codewiki (pinned by the repository configuration)" - ;; - auto) - echo "Stage 2 engine: auto (CodeWiki supported: $CODEWIKI_SUPPORTED)" - ;; - *) - echo "::error title=Invalid stage 2 engine::engine is '$CODEWIKI_ENGINE'; expected auto, codewiki or claude." - exit 1 - ;; - esac - - # Output for use by subsequent steps - set_output "primary_language" "$PRIMARY_LANG" - set_output "codewiki_supported" "$CODEWIKI_SUPPORTED" - set_output "file_count" "$MAX_COUNT" - - # ========================================================================= - # VALIDATE AND PARSE THE RUN CONFIGURATION - # 1. Every required parameter is present - # 2. Each JSON configuration is parsed into individual variables (the JSON - # grouping keeps the dispatch under GitHub's 25-input limit) - # 3. The parsed values are valid, then exported to GITHUB_ENV - # ========================================================================= - - name: Validate and parse run configuration - run: | - # 1. Required parameters ------------------------------------------- - VALIDATION_FAILED=0 - - # Core parameters - [ -z "$RUN_ID" ] && echo "::error title=Missing parameter::RUN_ID" && VALIDATION_FAILED=1 - [ -z "$REPO_ID" ] && echo "::error title=Missing parameter::REPO_ID" && VALIDATION_FAILED=1 - [ -z "$HUB_BASE_URL" ] && echo "::error title=Missing parameter::HUB_BASE_URL" && VALIDATION_FAILED=1 - - [ -z "$STAGES" ] && echo "::error title=Missing parameter::STAGES" && VALIDATION_FAILED=1 - [ -z "$CLAUDE_MODEL" ] && echo "::error title=Missing parameter::CLAUDE_MODEL" && VALIDATION_FAILED=1 - - # JSON parameters - [ -z "$CODEWIKI_CONFIG_JSON" ] && echo "::error title=Missing parameter::CODEWIKI_CONFIG_JSON" && VALIDATION_FAILED=1 - [ -z "$OUTPUT_PATHS_JSON" ] && echo "::error title=Missing parameter::OUTPUT_PATHS_JSON" && VALIDATION_FAILED=1 - [ -z "$README_CONFIG_JSON" ] && echo "::error title=Missing parameter::README_CONFIG_JSON" && VALIDATION_FAILED=1 - [ -z "$YOUTUBE_CONFIG_JSON" ] && echo "::error title=Missing parameter::YOUTUBE_CONFIG_JSON" && VALIDATION_FAILED=1 - - if [ $VALIDATION_FAILED -eq 1 ]; then - echo "::error title=Invalid run configuration::Required parameters are missing (annotated above). The hub sends every one of them; re-run \"Setup Workflow\" if this repository's workflow is out of date." - exit 1 - fi - - echo "Required parameters: present" - - # 2a. CodeWiki configuration (nested JSON) --------------------------- - echo "::group::CodeWiki configuration" - - # Only the keys with a real consumer are extracted here โ€” the per-phase - # base_url / api_version / temperature / temperature_supported are read - # directly from CODEWIKI_CONFIG_JSON by configure_codewiki_from_json, which - # is the single place that builds the `codewiki config set` command. They - # used to be parsed here as well and exported to $GITHUB_ENV, where nothing - # read them. - # - # No `// default` fallbacks anywhere below. The hub deep-merges every JSON - # param against the params SSOT before dispatch, so an absent key is a real - # bug โ€” and a fallback here would silently win over the SSOT, which is how - # docs/architecture and docs/reference/architecture drifted apart. - # `require_json_key` / `optional_json_key` come from workflow-helpers.sh. - source /tmp/workflow-helpers.sh - CW_JSON="$CODEWIKI_CONFIG_JSON" - - # Parse nested cluster config - CODEWIKI_CLUSTER_PROVIDER=$(require_json_key "$CW_JSON" '.cluster.provider' 'cluster provider') || exit 1 - CODEWIKI_CLUSTER_MODEL=$(require_json_key "$CW_JSON" '.cluster.model' 'cluster model') || exit 1 - CODEWIKI_CLUSTER_MAX_TOKENS=$(require_json_key "$CW_JSON" '.cluster.max_tokens' 'cluster max_tokens') || exit 1 - CODEWIKI_CLUSTER_MAX_TOKEN_FIELD=$(require_json_key "$CW_JSON" '.cluster.max_token_field' 'cluster max_token_field') || exit 1 - # Nullable by design: api_version is null for every OpenAI model. - - # Parse nested generation config - CODEWIKI_GENERATION_PROVIDER=$(require_json_key "$CW_JSON" '.generation.provider' 'generation provider') || exit 1 - CODEWIKI_GENERATION_MODEL=$(require_json_key "$CW_JSON" '.generation.model' 'generation model') || exit 1 - CODEWIKI_GENERATION_MAX_TOKENS=$(require_json_key "$CW_JSON" '.generation.max_tokens' 'generation max_tokens') || exit 1 - CODEWIKI_GENERATION_MAX_TOKEN_FIELD=$(require_json_key "$CW_JSON" '.generation.max_token_field' 'generation max_token_field') || exit 1 - - # Parse nested fallback config - CODEWIKI_FALLBACK_PROVIDER=$(require_json_key "$CW_JSON" '.fallback.provider' 'fallback provider') || exit 1 - CODEWIKI_FALLBACK_MODEL=$(require_json_key "$CW_JSON" '.fallback.model' 'fallback model') || exit 1 - CODEWIKI_FALLBACK_MAX_TOKENS=$(require_json_key "$CW_JSON" '.fallback.max_tokens' 'fallback max_tokens') || exit 1 - CODEWIKI_FALLBACK_MAX_TOKEN_FIELD=$(require_json_key "$CW_JSON" '.fallback.max_token_field' 'fallback max_token_field') || exit 1 - - # Parse top-level config. - # max_files_per_module is LIVE: this value reaches CodeWiki through the - # job-scoped $GITHUB_ENV write below, and upstream reads it in its - # empty-module-tree branch โ€” the branch Go/Rust/HCL repos land in. - CODEWIKI_MAX_FILES_PER_MODULE=$(require_json_key "$CW_JSON" '.max_files_per_module' 'max_files_per_module') || exit 1 - CODEWIKI_MAX_DEPTH=$(require_json_key "$CW_JSON" '.max_depth' 'max_depth') || exit 1 - CODEWIKI_REPO=$(require_json_key "$CW_JSON" '.repo' 'codewiki repo url') || exit 1 - - echo "Cluster (phase 2): $CODEWIKI_CLUSTER_PROVIDER / $CODEWIKI_CLUSTER_MODEL (${CODEWIKI_CLUSTER_MAX_TOKEN_FIELD}, ${CODEWIKI_CLUSTER_MAX_TOKENS} tokens)" - echo "Generation (phase 3+): $CODEWIKI_GENERATION_PROVIDER / $CODEWIKI_GENERATION_MODEL (${CODEWIKI_GENERATION_MAX_TOKEN_FIELD}, ${CODEWIKI_GENERATION_MAX_TOKENS} tokens)" - echo "Fallback: $CODEWIKI_FALLBACK_PROVIDER / $CODEWIKI_FALLBACK_MODEL (${CODEWIKI_FALLBACK_MAX_TOKEN_FIELD}, ${CODEWIKI_FALLBACK_MAX_TOKENS} tokens)" - echo "Max depth: $CODEWIKI_MAX_DEPTH" - echo "Max files per module: $CODEWIKI_MAX_FILES_PER_MODULE" - echo "::endgroup::" - - # 2b. YouTube configuration ------------------------------------------ - echo "::group::YouTube configuration" - - # Parse YouTube config from JSONB (channels only - API key from secrets) - # channels is legitimately optional: no channels == feature off - YOUTUBE_CHANNELS=$(echo "$YOUTUBE_CONFIG_JSON" | jq -c '.channels // []') - - # YouTube is enabled if channels array has items - YOUTUBE_ENABLED=$(echo "$YOUTUBE_CHANNELS" | jq -r 'if length > 0 then "true" else "false" end') - - echo "Enabled: $YOUTUBE_ENABLED (from the channel count)" - echo "Channels: $YOUTUBE_CHANNELS" - echo "API key: secrets.YOUTUBE_API_KEY (never stored on the hub)" - echo "::endgroup::" - - # 2c. README logo configuration -------------------------------------- - echo "::group::README logo configuration" - - # Parse README config from JSONB - README_LOGO_DARK=$(optional_json_key "$README_CONFIG_JSON" '.logo_dark') - README_LOGO_LIGHT=$(optional_json_key "$README_CONFIG_JSON" '.logo_light') - README_LOGO_ALT=$(require_json_key "$README_CONFIG_JSON" '.logo_alt' 'readme logo alt') || exit 1 - - echo "Dark logo: ${README_LOGO_DARK:-'(not set)'}" - echo "Light logo: ${README_LOGO_LIGHT:-'(not set)'}" - echo "Alt text: $README_LOGO_ALT" - echo "::endgroup::" - - # 2d. Output paths --------------------------------------------------- - echo "::group::Output paths" - - # Same fail-loud contract as the codewiki block. These fallbacks were the - # last surviving copy of the five paths, and `.reference` still said - # docs/architecture while the SSOT said docs/reference/architecture โ€” the - # very drift the SSOT was created to end. - OP_JSON="$OUTPUT_PATHS_JSON" - DOCS_OUTPUT_PATH=$(require_json_key "$OP_JSON" '.docs' 'docs output path') || exit 1 - REFERENCE_OUTPUT_PATH=$(require_json_key "$OP_JSON" '.reference' 'reference output path') || exit 1 - DIAGRAMS_OUTPUT_PATH=$(require_json_key "$OP_JSON" '.diagrams' 'diagrams output path') || exit 1 - GETTING_STARTED_OUTPUT_PATH=$(require_json_key "$OP_JSON" '.getting_started' 'getting-started output path') || exit 1 - DEVELOPMENT_OUTPUT_PATH=$(require_json_key "$OP_JSON" '.development' 'development output path') || exit 1 - - echo "Docs root: $DOCS_OUTPUT_PATH" - echo "Reference: $REFERENCE_OUTPUT_PATH" - echo "Diagrams: $DIAGRAMS_OUTPUT_PATH" - echo "Getting started: $GETTING_STARTED_OUTPUT_PATH" - echo "Development: $DEVELOPMENT_OUTPUT_PATH" - echo "::endgroup::" - - # 2e. Custom instructions and external repositories ------------------ - echo "::group::Custom instructions and external repositories" - - # Repository context (prevents invented GitHub URLs in every prompt) - # Extract repository information from GitHub context - GITHUB_REPO="${{ github.repository }}" - GITHUB_OWNER="${{ github.repository_owner }}" - GITHUB_SERVER="${{ github.server_url }}" - GITHUB_REPO_NAME=$(echo "$GITHUB_REPO" | cut -d'/' -f2) - GITHUB_REPO_URL="${GITHUB_SERVER}/${GITHUB_REPO}" - - # Build repository context section (injected into ALL AI prompts) - # Use printf for multi-line string (avoids YAML parsing issues with heredoc) - printf -v REPOSITORY_CONTEXT '%s\n' \ - '## REPOSITORY CONTEXT - GROUND TRUTH' \ - '' \ - '**CRITICAL:** This section provides the ACTUAL repository information. You MUST use these exact values when constructing GitHub URLs.' \ - '' \ - "- **Repository:** ${GITHUB_REPO}" \ - "- **Owner:** ${GITHUB_OWNER}" \ - "- **Repository Name:** ${GITHUB_REPO_NAME}" \ - "- **Repository URL:** ${GITHUB_REPO_URL}" \ - "- **Server:** ${GITHUB_SERVER}" \ - '' \ - '**MANDATORY RULES FOR GITHUB URLS:**' \ - "1. ALWAYS use the exact repository path: \`${GITHUB_REPO}\`" \ - '2. NEVER use placeholder URLs like "your-org", "example-org", or "mycompany"' \ - '3. NEVER infer repository owner from file contents or dependencies' \ - '4. NEVER use upstream/parent repository URLs (if this is a fork, use the fork URL)' \ - "5. When linking to code: \`${GITHUB_REPO_URL}/blob/main/path/to/file\`" \ - "6. When linking to clone: \`git clone ${GITHUB_REPO_URL}.git\`" \ - "7. When linking to issues/PRs: \`${GITHUB_REPO_URL}/issues\` or \`${GITHUB_REPO_URL}/pulls\`" \ - "8. When linking to releases: \`${GITHUB_REPO_URL}/releases\`" \ - '' \ - '**If you find yourself writing a GitHub URL, verify it matches the Repository URL above.**' \ - "**ESPECIALLY IN README.md and tutorials - All GitHub URLs MUST use ${GITHUB_REPO}**" \ - '' \ - '---' - - echo "Repository: $GITHUB_REPO" - echo "Repository URL: $GITHUB_REPO_URL" - - # Repository context goes in front of the custom instructions - # Custom instructions come as plain text from user - # Read from the environment rather than interpolating into single quotes. - # GitHub Actions expression substitution runs BEFORE bash parses the line, so a - # single apostrophe anywhere in an admin's instructions used to terminate the - # string and kill the step with a syntax error. NOTE: never write a literal - # empty GitHub expression in this run block, even inside a comment โ€” Actions - # evaluates it pre-bash and the whole workflow fails to parse. - USER_CUSTOM_INSTRUCTIONS="$CUSTOM_REPO_INSTRUCTIONS" - - # Combine repository context + user custom instructions - # Repository context goes FIRST (highest priority in prompts) - if [ -n "$USER_CUSTOM_INSTRUCTIONS" ]; then - CUSTOM_INSTRUCTIONS="${REPOSITORY_CONTEXT}"$'\n\n'"${USER_CUSTOM_INSTRUCTIONS}" - else - CUSTOM_INSTRUCTIONS="$REPOSITORY_CONTEXT" - fi - - # External repos come as separate JSON array parameter - - EXTERNAL_REPOS_COUNT=$(echo "$EXTERNAL_REPOS" | jq '. | length' 2>/dev/null || echo "0") - echo "Repository context: ${#REPOSITORY_CONTEXT} chars" - echo "Custom instructions: ${#USER_CUSTOM_INSTRUCTIONS} chars" - echo "Combined prompt block: ${#CUSTOM_INSTRUCTIONS} chars" - echo "External repositories: $EXTERNAL_REPOS_COUNT" - echo "Combined prompt block, first 300 chars:" - echo "${CUSTOM_INSTRUCTIONS:0:300}..." - echo "::endgroup::" - - # 3. Parsed values --------------------------------------------------- - - # Validate providers - if [[ ! "$CODEWIKI_CLUSTER_PROVIDER" =~ ^(anthropic|openai)$ ]]; then - echo "::error title=Invalid CodeWiki configuration::cluster provider is '$CODEWIKI_CLUSTER_PROVIDER'; expected anthropic or openai." - exit 1 - fi - - if [[ ! "$CODEWIKI_GENERATION_PROVIDER" =~ ^(anthropic|openai)$ ]]; then - echo "::error title=Invalid CodeWiki configuration::generation provider is '$CODEWIKI_GENERATION_PROVIDER'; expected anthropic or openai." - exit 1 - fi - - if [[ ! "$CODEWIKI_FALLBACK_PROVIDER" =~ ^(anthropic|openai)$ ]]; then - echo "::error title=Invalid CodeWiki configuration::fallback provider is '$CODEWIKI_FALLBACK_PROVIDER'; expected anthropic or openai." - exit 1 - fi - - # Model names are not empty - [ -z "$CODEWIKI_CLUSTER_MODEL" ] && echo "::error title=Invalid CodeWiki configuration::cluster model is empty." && exit 1 - [ -z "$CODEWIKI_GENERATION_MODEL" ] && echo "::error title=Invalid CodeWiki configuration::generation model is empty." && exit 1 - [ -z "$CODEWIKI_FALLBACK_MODEL" ] && echo "::error title=Invalid CodeWiki configuration::fallback model is empty." && exit 1 - - # Numeric values - [[ ! "$CODEWIKI_CLUSTER_MAX_TOKENS" =~ ^[0-9]+$ ]] && echo "::error title=Invalid CodeWiki configuration::cluster max_tokens is '$CODEWIKI_CLUSTER_MAX_TOKENS', not a number." && exit 1 - [[ ! "$CODEWIKI_GENERATION_MAX_TOKENS" =~ ^[0-9]+$ ]] && echo "::error title=Invalid CodeWiki configuration::generation max_tokens is '$CODEWIKI_GENERATION_MAX_TOKENS', not a number." && exit 1 - [[ ! "$CODEWIKI_FALLBACK_MAX_TOKENS" =~ ^[0-9]+$ ]] && echo "::error title=Invalid CodeWiki configuration::fallback max_tokens is '$CODEWIKI_FALLBACK_MAX_TOKENS', not a number." && exit 1 - [[ ! "$CODEWIKI_MAX_DEPTH" =~ ^[0-9]+$ ]] && echo "::error title=Invalid CodeWiki configuration::max_depth is '$CODEWIKI_MAX_DEPTH', not a number." && exit 1 - [[ ! "$CODEWIKI_MAX_FILES_PER_MODULE" =~ ^[0-9]+$ ]] && echo "::error title=Invalid CodeWiki configuration::max_files_per_module is '$CODEWIKI_MAX_FILES_PER_MODULE', not a number." && exit 1 - - echo "Parsed values: valid" - - # 4. Export to GITHUB_ENV (every later step reads these) ------------- - - # CodeWiki config (31 vars: cluster=9, generation=10, fallback=9, shared=3) - echo "CODEWIKI_CLUSTER_PROVIDER=$CODEWIKI_CLUSTER_PROVIDER" >> $GITHUB_ENV - echo "CODEWIKI_CLUSTER_MODEL=$CODEWIKI_CLUSTER_MODEL" >> $GITHUB_ENV - echo "CODEWIKI_CLUSTER_MAX_TOKENS=$CODEWIKI_CLUSTER_MAX_TOKENS" >> $GITHUB_ENV - echo "CODEWIKI_CLUSTER_MAX_TOKEN_FIELD=$CODEWIKI_CLUSTER_MAX_TOKEN_FIELD" >> $GITHUB_ENV - echo "CODEWIKI_GENERATION_PROVIDER=$CODEWIKI_GENERATION_PROVIDER" >> $GITHUB_ENV - echo "CODEWIKI_GENERATION_MODEL=$CODEWIKI_GENERATION_MODEL" >> $GITHUB_ENV - echo "CODEWIKI_GENERATION_MAX_TOKENS=$CODEWIKI_GENERATION_MAX_TOKENS" >> $GITHUB_ENV - echo "CODEWIKI_GENERATION_MAX_TOKEN_FIELD=$CODEWIKI_GENERATION_MAX_TOKEN_FIELD" >> $GITHUB_ENV - echo "CODEWIKI_FALLBACK_PROVIDER=$CODEWIKI_FALLBACK_PROVIDER" >> $GITHUB_ENV - echo "CODEWIKI_FALLBACK_MODEL=$CODEWIKI_FALLBACK_MODEL" >> $GITHUB_ENV - echo "CODEWIKI_FALLBACK_MAX_TOKENS=$CODEWIKI_FALLBACK_MAX_TOKENS" >> $GITHUB_ENV - echo "CODEWIKI_FALLBACK_MAX_TOKEN_FIELD=$CODEWIKI_FALLBACK_MAX_TOKEN_FIELD" >> $GITHUB_ENV - echo "CODEWIKI_MAX_FILES_PER_MODULE=$CODEWIKI_MAX_FILES_PER_MODULE" >> $GITHUB_ENV - echo "CODEWIKI_MAX_DEPTH=$CODEWIKI_MAX_DEPTH" >> $GITHUB_ENV - echo "CODEWIKI_REPO=$CODEWIKI_REPO" >> $GITHUB_ENV - - # YouTube config (2 vars - API key from secrets, not exported here) - echo "YOUTUBE_ENABLED=$YOUTUBE_ENABLED" >> $GITHUB_ENV - echo "YOUTUBE_CHANNELS=$YOUTUBE_CHANNELS" >> $GITHUB_ENV - - # README config (3 vars) - echo "README_LOGO_DARK=$README_LOGO_DARK" >> $GITHUB_ENV - echo "README_LOGO_LIGHT=$README_LOGO_LIGHT" >> $GITHUB_ENV - echo "README_LOGO_ALT=$README_LOGO_ALT" >> $GITHUB_ENV - - # Output paths (5 vars) - echo "DOCS_OUTPUT_PATH=$DOCS_OUTPUT_PATH" >> $GITHUB_ENV - echo "REFERENCE_OUTPUT_PATH=$REFERENCE_OUTPUT_PATH" >> $GITHUB_ENV - echo "DIAGRAMS_OUTPUT_PATH=$DIAGRAMS_OUTPUT_PATH" >> $GITHUB_ENV - echo "GETTING_STARTED_OUTPUT_PATH=$GETTING_STARTED_OUTPUT_PATH" >> $GITHUB_ENV - echo "DEVELOPMENT_OUTPUT_PATH=$DEVELOPMENT_OUTPUT_PATH" >> $GITHUB_ENV - - # Custom instructions (2 vars) - echo "CUSTOM_INSTRUCTIONS<> $GITHUB_ENV - echo "$CUSTOM_INSTRUCTIONS" >> $GITHUB_ENV - echo "EOF" >> $GITHUB_ENV - echo "EXTERNAL_REPOS=$EXTERNAL_REPOS" >> $GITHUB_ENV - - echo "Run configuration validated and exported: 15 CodeWiki, 2 YouTube, 3 README, 5 output paths, 2 instruction values" - - # ========================================================================= - # SOURCE FILE DISCOVERY (the one list every stage reads) - # Discovers SOURCE CODE files and optionally DELETES everything else. - # When SOURCE_FILES_LIMIT > 0: - # 1. Keeps only N source files (.ts, .java, .py, etc.) - # 2. DELETES ALL other files in the repo (aggressive cleanup) - # Generated docs (inline .md) are created AFTER this step, so not affected. - # ========================================================================= - - name: Discover source files - id: discover_files - env: - SOURCE_FILES_LIMIT: ${{ env.SOURCE_FILES_LIMIT }} - # DOCS_OUTPUT_PATH is needed below so the find can exclude the - # generated docs tree (deleted by the docs-removal step that runs - # AFTER discovery but BEFORE Stage 1 โ€” source-extension files - # under docs/ would otherwise be enumerated, then deleted, then - # cause ENOENT in Stage 1). - DOCS_OUTPUT_PATH: ${{ env.DOCS_OUTPUT_PATH }} - run: | - source /tmp/workflow-helpers.sh - - echo "Discovering source files in the repository (.) and its dependencies (../deps/)" - - # File paths (all in /tmp to avoid accidental commits) - SOURCE_FILES_LIST="/tmp/.doc-orchestrator-source-files.txt" - ALL_SOURCE_TEMP="/tmp/all_source_files_discovered.txt" - ALL_FILES_TEMP="/tmp/all_files_in_repo.txt" - FILES_TO_DELETE="/tmp/files_to_delete.txt" - - # 1. Find SOURCE CODE files only (with standard exclusions) - # IMPORTANT: exclude $DOCS_OUTPUT_PATH/* โ€” the docs-removal step below - # does `rm -rf $DOCS_OUTPUT_PATH` BEFORE Stage 1 reads this list. If - # any source-extension file lives under the docs tree (Doxygen's - # `docs/doxygen/documentation.h`, Sphinx `_extensions/*.py`, etc.), - # it would be enumerated here, then deleted, then Stage 1 hits - # ENOENT trying to read it. This is the source-discovery / - # docs-removal / Stage-1 ordering bug โ€” exclusion is the targeted fix. - # The list comes from ci-source.mjs: the served source extensions, the - # served never-source directories, tests skipped by name โ€” the same rule - # the language detection above counted with. - node /tmp/ci-source.mjs list "$ALL_SOURCE_TEMP" --exclude "./$DOCS_OUTPUT_PATH" > /dev/null || { echo "::error title=Source discovery failed::ci-source.mjs list exited non-zero."; exit 1; } - - TOTAL_SOURCE=$(wc -l < "$ALL_SOURCE_TEMP" | tr -d ' ') - FILE_LIMIT="${SOURCE_FILES_LIMIT:-0}" - - echo "Source files found: $TOTAL_SOURCE" - - # Apply limit: keep N source files, DELETE EVERYTHING ELSE - if [ "$FILE_LIMIT" -gt 0 ]; then - echo "::warning title=Source file limit active::SOURCE_FILES_LIMIT=$FILE_LIMIT. Keeping $FILE_LIMIT source files and deleting every other file in the checkout (a debugging setting)." - echo "::group::Apply the source file limit" - - # Keep first N source files - head -n "$FILE_LIMIT" "$ALL_SOURCE_TEMP" > "$SOURCE_FILES_LIST" - KEEPING=$(wc -l < "$SOURCE_FILES_LIST" | tr -d ' ') - - # 2. Find ALL files in the repo (except .git and workflow temp files) - find . ../deps 2>/dev/null -type f \ - -not -path "*/.git/*" \ - -not -path "*/.git" \ - -not -name ".doc-orchestrator-*" \ - -not -name ".doc-stage*" \ - | sort > "$ALL_FILES_TEMP" - - TOTAL_FILES=$(wc -l < "$ALL_FILES_TEMP" | tr -d ' ') - echo "Files in the checkout: $TOTAL_FILES" - - # Build delete list: ALL files EXCEPT the ones we're keeping - # Also preserve workflow temp files (.doc-orchestrator-*, .doc-stage*) - > "$FILES_TO_DELETE" - while IFS= read -r file; do - # Skip workflow temp files we need to preserve - case "$file" in - ./.doc-orchestrator-*|./.doc-stage*) continue ;; - esac - # Check if this file is in our keep list - if ! grep -qxF "$file" "$SOURCE_FILES_LIST" 2>/dev/null; then - echo "$file" >> "$FILES_TO_DELETE" - fi - done < "$ALL_FILES_TEMP" - - DELETE_COUNT=$(wc -l < "$FILES_TO_DELETE" | tr -d ' ') - echo "Files to delete: $DELETE_COUNT" - - # Delete all files NOT in the keep list - DELETED_COUNT=0 - while IFS= read -r file_to_delete; do - if [ -f "$file_to_delete" ]; then - rm -f "$file_to_delete" - DELETED_COUNT=$((DELETED_COUNT + 1)) - fi - done < "$FILES_TO_DELETE" - - echo "Kept $KEEPING source files, deleted $DELETED_COUNT files" - - rm -f "$FILES_TO_DELETE" "$ALL_FILES_TEMP" - - # Aggressively prune directories (including those with only dotfiles) - PRUNED_COUNT=0 - - # First, delete all dotfiles except in .git and workflow temp files (they prevent dir deletion) - find . -type f -name ".*" \ - -not -path "*/.git/*" \ - -not -name ".doc-orchestrator-*" \ - -not -name ".doc-stage*" \ - -delete 2>/dev/null || true - - # Multiple passes to handle nested empty directories - for i in 1 2 3 4 5 6 7 8 9 10; do - PASS_COUNT=0 - while IFS= read -r empty_dir; do - if [ -d "$empty_dir" ] && [ -z "$(ls -A "$empty_dir" 2>/dev/null)" ]; then - rmdir "$empty_dir" 2>/dev/null && PASS_COUNT=$((PASS_COUNT + 1)) - fi - done < <(find . -type d -empty 2>/dev/null | grep -v "^.$" | grep -v ".git") - PRUNED_COUNT=$((PRUNED_COUNT + PASS_COUNT)) - [ "$PASS_COUNT" -eq 0 ] && break - done - - if [ "$PRUNED_COUNT" -gt 0 ]; then - echo "Pruned $PRUNED_COUNT empty directories" - fi - - # Show what's left - echo "Remaining directories (first 20):" - find . -type d -not -path "*/.git/*" -not -path "*/.git" | head -20 - echo "::endgroup::" - else - # No limit - keep all source files - cp "$ALL_SOURCE_TEMP" "$SOURCE_FILES_LIST" - fi - - # Cleanup temp file - rm -f "$ALL_SOURCE_TEMP" - - # Count remaining files - FILE_COUNT=$(count_source_files "$SOURCE_FILES_LIST") - MAIN_COUNT=$(count_main_repo_files "$SOURCE_FILES_LIST") - DEPS_COUNT=$(count_dependency_files "$SOURCE_FILES_LIST") - - echo "Source files to document: $FILE_COUNT ($MAIN_COUNT in this repository, $DEPS_COUNT in dependencies)" - - echo "::group::Source files by language" - node /tmp/ci-source.mjs breakdown "$SOURCE_FILES_LIST" | sed 's/^/ /' - echo "::endgroup::" - - echo "::group::Sample source files (first 20)" - head -20 "$SOURCE_FILES_LIST" | sed 's/^/ /' - echo "::endgroup::" - - # Output for use by subsequent steps - set_output "source_file_count" "$FILE_COUNT" - set_output "source_files_list" "$SOURCE_FILES_LIST" - - # ========================================================================= - # PULL REQUEST FIRST (progressive pull request) - # The branch and the pull request exist before any stage runs, so each - # stage commits its results the moment it finishes. A timeout loses at - # most the stage in flight. - # ========================================================================= - - name: Derive branch name - id: branch-name-early - run: | - source /tmp/workflow-helpers.sh - # Replace colons and other invalid chars with hyphens for git branch name - SAFE_RUN_ID=$(echo "$RUN_ID" | sed 's/[:]/-/g' | sed 's/[^a-zA-Z0-9._-]/-/g') - set_output "safe_run_id" "$SAFE_RUN_ID" - echo "Run ID for the branch name: $SAFE_RUN_ID" - - - name: Create docs branch - id: create-pr-branch - run: | - source /tmp/workflow-helpers.sh - - # Use sanitized run ID for branch name. Branch, PR title, status file and - # commit messages all carry the product name: ๐Ÿฆฉ Flamingo Code Documentation. - SAFE_RUN_ID="${{ steps.branch-name-early.outputs.safe_run_id }}" - BRANCH_NAME="docs/flamingo-ai-technical-writer-$SAFE_RUN_ID" - - # Configure git - git config user.name "github-actions[bot]" - git config user.email "github-actions[bot]@users.noreply.github.com" - - # Create and push empty branch - git checkout -b "$BRANCH_NAME" - - # Create initial commit to enable PR creation - echo "# ๐Ÿฆฉ Flamingo Code Documentation: Started" > .flamingo-ai-technical-writer-status.md - echo "" >> .flamingo-ai-technical-writer-status.md - echo "Run ID: $SAFE_RUN_ID" >> .flamingo-ai-technical-writer-status.md - echo "Status: In Progress" >> .flamingo-ai-technical-writer-status.md - echo "Started: $(date -u +"%Y-%m-%d %H:%M:%S UTC")" >> .flamingo-ai-technical-writer-status.md - # -f: the status file is a hidden dot-md that many target repos' .gitignore - # patterns (e.g. `.*` / `*status*`) cover โ€” without -f, `git add` fails the - # step. It's removed again in "Clean up temporary files" before the PR. - git add -f .flamingo-ai-technical-writer-status.md - git commit -m "docs: Initialize ๐Ÿฆฉ Flamingo Code Documentation run [skip ci]" - git push -u origin "$BRANCH_NAME" - - # Store branch name for later steps - set_output "branch_name" "$BRANCH_NAME" - - echo "Docs branch pushed: $BRANCH_NAME" - - - name: Ensure pull request labels exist - env: - GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} - run: | - # Labels used for ๐Ÿฆฉ Flamingo Code Documentation PRs - LABELS=( - "documentation:A label for documentation-related PRs:#0075ca" - "automated:PRs created by automation/bots:#ededed" - "in-progress:Work in progress - not ready for merge:#fbca04" - ) - - for LABEL_DEF in "${LABELS[@]}"; do - LABEL_NAME=$(echo "$LABEL_DEF" | cut -d: -f1) - LABEL_DESC=$(echo "$LABEL_DEF" | cut -d: -f2) - LABEL_COLOR=$(echo "$LABEL_DEF" | cut -d: -f3 | sed 's/#//') - - # Check if label exists - if gh label list --json name --jq '.[].name' | grep -q "^${LABEL_NAME}$"; then - echo "Label exists: $LABEL_NAME" - else - echo "Creating label: $LABEL_NAME" - gh label create "$LABEL_NAME" \ - --description "$LABEL_DESC" \ - --color "$LABEL_COLOR" || true - fi - done - - - name: Open pull request - id: create-initial-pr - env: - GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} - BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }} - RUN_ID_VAR: ${{ env.RUN_ID }} - REPO_NAME: ${{ github.repository }} - DEFAULT_BRANCH: ${{ github.event.repository.default_branch }} - run: | - source /tmp/workflow-helpers.sh - - # Create PR body in a temp file (avoiding YAML parsing issues) - { - echo "๐Ÿฆฉ Flamingo Code Documentation: In Progress" - echo "" - echo "Run ID: $RUN_ID_VAR" - echo "Status: Running..." - echo "" - echo "This PR will be updated as each documentation stage completes." - echo "" - echo "Progress" - echo "- Stage 1 Inline Documentation - Starting..." - echo "- Stage 2 Architecture Analysis - Pending" - echo "- Stage 3 Tutorial Generation - Pending" - echo "- Stage 4 Repository Documentation - Pending" - echo "" - echo "Generated by ๐Ÿฆฉ Flamingo Code Documentation" - } > /tmp/pr-body.md - - # Create PR with gh CLI (works with existing branches) - PR_URL=$(gh pr create \ - --base "$DEFAULT_BRANCH" \ - --head "$BRANCH_NAME" \ - --title "[IN PROGRESS] ๐Ÿฆฉ Flamingo Code Documentation" \ - --body-file /tmp/pr-body.md \ - --label "documentation,automated,in-progress") - - # Extract PR number from URL - PR_NUMBER=$(echo "$PR_URL" | grep -oE '[0-9]+$') - - echo "Pull request #$PR_NUMBER opened: $PR_URL" - - # Set outputs for later steps - set_output "pull-request-url" "$PR_URL" - set_output "pull-request-number" "$PR_NUMBER" - - # ========================================================================= - # REMOVE DOCS THIS RUN REGENERATES - # Deletes the documentation the configured stages own before they run, so - # the result carries no orphaned files. A subtree whose stage then - # produces nothing is put back by "Restore docs no stage regenerated". - # ========================================================================= - - name: Remove docs this run regenerates - env: - DOCS_OUTPUT_PATH: ${{ env.DOCS_OUTPUT_PATH }} - REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }} - DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }} - GETTING_STARTED_OUTPUT_PATH: ${{ env.GETTING_STARTED_OUTPUT_PATH }} - DEVELOPMENT_OUTPUT_PATH: ${{ env.DEVELOPMENT_OUTPUT_PATH }} - BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }} - run: | - source /tmp/workflow-helpers.sh - - # Recorded so a stage that ends up producing nothing can restore what was - # deleted on its behalf. Without it, a skipped stage turns the pull request - # into a net DELETION of existing documentation. - PRE_CLEAN_SHA=$(git rev-parse HEAD) - echo "PRE_CLEAN_SHA=$PRE_CLEAN_SHA" >> $GITHUB_ENV - echo "Commit before removal: $PRE_CLEAN_SHA" - - # SCOPED TO THE CONFIGURED STAGES. - # - # This used to `rm -rf $DOCS_OUTPUT_PATH` unconditionally. That was safe only - # while every repo ran all four stages. With per-repo `stages`, wiping the - # whole tree when Stage 2 is disabled means the reference architecture and - # diagrams are deleted and never rebuilt โ€” the pull request becomes a net - # DELETION of existing documentation. - CLEAN_TARGETS=() - if [[ "$STAGES" == *"codewiki"* ]]; then - CLEAN_TARGETS+=("$REFERENCE_OUTPUT_PATH" "$DIAGRAMS_OUTPUT_PATH") - fi - if [[ "$STAGES" == *"tutorials"* ]]; then - CLEAN_TARGETS+=("$GETTING_STARTED_OUTPUT_PATH" "$DEVELOPMENT_OUTPUT_PATH") - fi - - # DELETE_TARGETS is what actually gets rm -rf'd; CLEAN_TARGETS is what the - # restore step keys PER-STAGE. They differ only for a full wipe: we delete - # the whole tree (so orphaned files from a previous layout โ€” e.g. synthetic - # module_N dirs left by an earlier clustering-failure run โ€” cannot survive) - # but still RECORD the per-stage subtrees, so the restore leaves each - # subtree deleted iff its OWN stage produced output. - # - # Recording the blanket DOCS_OUTPUT_PATH instead (the old behaviour) made the - # restore treat docs/ as a single unit that is "safe to leave deleted" only - # once ALL FOUR stages complete โ€” but the restore runs right after Stage 2, - # so Stage 3/4 are never 'completed' yet, and it restored the ENTIRE pre-clean - # tree every time, undoing the wipe and resurrecting the orphaned module_N docs. - DELETE_TARGETS=("${CLEAN_TARGETS[@]}") - FULL_WIPE=false - SELECTED_COUNT=$(echo "$STAGES" | tr ',' '\n' | grep -c .) - if [ "$SELECTED_COUNT" -ge "${STAGE_COUNT:-4}" ]; then - FULL_WIPE=true - DELETE_TARGETS=("$DOCS_OUTPUT_PATH") - # Record every stage-owned subtree (NOT the blanket docs/) so the restore - # keys each subtree on its own stage instead of the all-four AND. - CLEAN_TARGETS=("$REFERENCE_OUTPUT_PATH" "$DIAGRAMS_OUTPUT_PATH" "$GETTING_STARTED_OUTPUT_PATH" "$DEVELOPMENT_OUTPUT_PATH") - fi - - echo "Stages: $STAGES" - echo "Full wipe: $FULL_WIPE" - echo "Targets: ${CLEAN_TARGETS[*]:-(none)}" - - # Recorded so the restore step iterates exactly what was removed โ€” the - # stage -> subtree map has ONE home, here. - { - echo "CLEAN_TARGETS_RECORD<<__EOT__" - for t in ${CLEAN_TARGETS[@]+"${CLEAN_TARGETS[@]}"}; do echo "$t"; done - echo "__EOT__" - } >> $GITHUB_ENV - - # Count files before deletion (for reporting). Iterate DELETE_TARGETS โ€” - # the actual rm list (blanket docs/ on a full wipe, per-stage subtrees - # otherwise) โ€” not the restore-record CLEAN_TARGETS. - DELETED_FILES=0 - for target in ${DELETE_TARGETS[@]+"${DELETE_TARGETS[@]}"}; do - if [ -d "$target" ]; then - TARGET_FILES=$(find "$target" -type f | wc -l | tr -d ' ') - DELETED_FILES=$((DELETED_FILES + TARGET_FILES)) - echo "Removing $target ($TARGET_FILES files)" - rm -rf "$target" - else - echo "Skipping $target (does not exist)" - fi - done - if [ ${#DELETE_TARGETS[@]} -eq 0 ]; then - echo "No configured stage owns a docs subtree; nothing to remove" - fi - - # Recreate base directory - mkdir -p "$DOCS_OUTPUT_PATH" - - # Commit the deletion to git (so it shows in PR) - if [ "$DELETED_FILES" -gt 0 ]; then - git add -A - - # Check if there are staged changes - STAGED_COUNT=$(git diff --cached --name-only | wc -l | tr -d ' ') - - if [ "$STAGED_COUNT" -gt 0 ]; then - git commit -m "chore(docs): Clean slate - remove all documentation ($DELETED_FILES files) [skip ci]" - git push origin "$BRANCH_NAME" - echo "Committed the removal of $STAGED_COUNT files" - else - echo "Nothing to commit (the docs were already absent)" - fi - fi - - - name: Set up Node.js - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0 - with: - node-version: '22' - - # ========================================================================= - # STAGE 0: CODE GRAPH (same build as the standalone code-graph job above) - # Tags the SOURCE branch head so the hub can render this run's - # ecosystem.md from the snapshot of the commit being documented. The hub - # promotes to `live` only when the source branch is the default branch. - # Never fatal: a graph failure costs cross-repo facts, not the docs run. - # ========================================================================= - # The command is CODE_GRAPH_INSTALL_COMMAND (lib/config/code-graph-workflow.ts). - - name: Install graph dependencies - continue-on-error: true - run: mkdir -p "$RUNNER_TEMP/code-graph-deps" && cd "$RUNNER_TEMP/code-graph-deps" && printf '%s' '{"name":"code-graph-deps","version":"1.0.0","private":true,"dependencies":{"web-tree-sitter":"0.27.0","@vscode/tree-sitter-wasm":"0.3.1","yaml":"2.9.1"}}' > package.json && printf '%s' '{"name":"code-graph-deps","version":"1.0.0","lockfileVersion":3,"requires":true,"packages":{"":{"name":"code-graph-deps","version":"1.0.0","dependencies":{"web-tree-sitter":"0.27.0","@vscode/tree-sitter-wasm":"0.3.1","yaml":"2.9.1"}},"node_modules/web-tree-sitter":{"version":"0.27.0","resolved":"https://registry.npmjs.org/web-tree-sitter/-/web-tree-sitter-0.27.0.tgz","integrity":"sha512-XK08gj6RwTMQatAG7uVRP8MunqotL/XC19vHgkSPKmELgbGPBj4ECvB8haHOUnyj6ls2B8t42UTro14zxGgAHg=="},"node_modules/@vscode/tree-sitter-wasm":{"version":"0.3.1","resolved":"https://registry.npmjs.org/@vscode/tree-sitter-wasm/-/tree-sitter-wasm-0.3.1.tgz","integrity":"sha512-RJFoomET6FajjG511fmQxeBQfU6M24a0aFZPqpid+ttIxanWf1VGytBG0UmsGjt07qmIPJS8U31D+aecuCucsQ=="},"node_modules/yaml":{"version":"2.9.1","resolved":"https://registry.npmjs.org/yaml/-/yaml-2.9.1.tgz","integrity":"sha512-3NxN8+78OdzbT7C/WjGsyfPAtJaN3FNDsWxv7Y7mcDsT/oOmgW8BpyQQFFBnvZE3j9Y2Sdz1ULFLezL7Eb2yFw=="}}}' > package-lock.json && npm ci --ignore-scripts --no-audit --no-fund && echo "CODE_GRAPH_DEPS_DIR=$RUNNER_TEMP/code-graph-deps" >> "$GITHUB_ENV" || { echo "::warning::graph dependencies failed their lockfile-enforced install; continuing without them"; rm -rf "$RUNNER_TEMP/code-graph-deps"; exit 1; } - - - name: Build and upload the code graph - id: graph - continue-on-error: true - env: - WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }} - CODE_GRAPH_DEPS_DIR: ${{ env.CODE_GRAPH_DEPS_DIR }} - GITHUB_REPOSITORY: ${{ github.repository }} - CODE_GRAPH_BRANCH: ${{ env.SOURCE_BRANCH }} - CODE_GRAPH_COMMIT_SHA: ${{ env.SOURCE_HEAD_SHA }} - run: node /tmp/code-graph-build.mjs - - # ========================================================================= - # STAGE 1: INLINE DOCS - # One hidden .md beside each source file, over the discovered file list. - # ========================================================================= - - name: Install stage 1 dependencies - if: contains(env.STAGES, 'inline-docs') - # Install generator deps in an ISOLATED tree under RUNNER_TEMP, NOT the target - # repo. npm resolves against an empty package.json here, so a target repo's own - # peer conflicts (e.g. react-accessible-accordion vs react 18) can never make this - # fail. No --legacy-peer-deps / --no-save band-aids. Generators find these via the - # NODE_PATH set on the generate step (RUNNER_TEMP/doc-orch-deps/node_modules). - run: | - mkdir -p "$RUNNER_TEMP/doc-orch-deps" && cd "$RUNNER_TEMP/doc-orch-deps" - npm init -y >/dev/null 2>&1 - npm install @anthropic-ai/sdk@0.115.0 zod@3.25.76 glob@13.0.6 - - - name: Generate inline docs (stage 1) - id: stage1 - if: contains(env.STAGES, 'inline-docs') - env: - # SECURITY: Pass secrets per-step with inline masking. - # No ANTHROPIC_API_KEY โ€” this stage calls Claude through the hub - # (/api/ci/claude), which the WEBHOOK_SECRET below authenticates. - WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }} - # Claude model SSOT โ€” see workflow env CLAUDE_MODEL block - CLAUDE_MODEL: ${{ env.CLAUDE_MODEL }} - DOCS_OUTPUT_PATH: ${{ env.DOCS_OUTPUT_PATH }} - # Stage timeout (in hours) - STAGE1_TIMEOUT_HOURS: ${{ env.STAGE1_TIMEOUT_HOURS }} - # Incremental commit + push every N generated docs (see workflow env) - STAGE1_PUSH_INTERVAL: ${{ env.STAGE1_PUSH_INTERVAL }} - BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }} - STAGE1_PUSH_MARKER: /tmp/stage1-progress-pushes - # Unified file discovery result (single source of truth) - SOURCE_FILES_LIST: ${{ steps.discover_files.outputs.source_files_list }} - SOURCE_FILE_COUNT: ${{ steps.discover_files.outputs.source_file_count }} - # NODE_PATH to find modules from /tmp/ scripts - NODE_PATH: ${{ runner.temp }}/doc-orch-deps/node_modules - # Custom AI Instructions (All Stages) - CUSTOM_REPO_INSTRUCTIONS: ${{ env.CUSTOM_INSTRUCTIONS }} - # External Repositories - EXTERNAL_REPOS: ${{ env.EXTERNAL_REPOS }} - run: | - source /tmp/workflow-helpers.sh - - echo "Source files: $SOURCE_FILE_COUNT (from $SOURCE_FILES_LIST)" - echo "Progress push: every $STAGE1_PUSH_INTERVAL generated docs, to $BRANCH_NAME" - - # Fresh marker: the generator appends one line per progress push, and the - # commit step below reads it to know work was already pushed. - rm -f "$STAGE1_PUSH_MARKER" - - # Script already downloaded to /tmp/ in setup step. run_stage records - # the outcome as stage1_status โ€” see its note in workflow-helpers.sh. - run_stage "Stage 1" "$STAGE1_TIMEOUT_HOURS" stage1_status node /tmp/generate-inline-docs.cjs - - # Count generated files (hidden .*.md files) - INLINE_DOCS=$(find . -name ".*.md" -newer .git -type f -not -path "./node_modules/*" -not -path "./.git/*" | wc -l) - set_output "stage1_files" "$INLINE_DOCS" - - # ========================================================================= - # COMMIT STAGE 1 RESULTS (progressive pull request) - # ========================================================================= - # `!= ''`, not `== 'completed'`: runs on a FAILED stage too โ€” see run_stage - # in workflow-helpers.sh (partial output is worth committing; the status - # is what reports the truth home). The build gate holds every run_stage - # commit step to this predicate. - - name: Commit and push stage 1 results - if: always() && steps.stage1.outputs.stage1_status != '' - id: commit-stage1 - env: - BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }} - STAGE_FILES: ${{ steps.stage1.outputs.stage1_files }} - STAGE1_PUSH_MARKER: /tmp/stage1-progress-pushes - run: | - source /tmp/workflow-helpers.sh - - # The generator already commits + pushes every N docs (STAGE1_PUSH_INTERVAL). - # Report how much landed that way; what's left here is the final partial batch. - if [ -s "$STAGE1_PUSH_MARKER" ]; then - PUSHED_BATCHES=$(wc -l < "$STAGE1_PUSH_MARKER" | tr -d ' ') - PUSHED_FILES=$(awk '{ sum += $1 } END { print sum + 0 }' "$STAGE1_PUSH_MARKER") - echo "Already pushed during generation: $PUSHED_FILES files in $PUSHED_BATCHES batches" - fi - - # Push any commits the generator made but could not push (transient push failure) - git push origin "HEAD:refs/heads/$BRANCH_NAME" 2>/dev/null || true - - # Stage all .md files generated by Stage 1 (hidden inline docs) - find . -name ".*.md" -type f \ - -not -path "./node_modules/*" \ - -not -path "./.git/*" \ - -exec git add -f {} \; 2>/dev/null || true - - # Check if there are changes - STAGED_COUNT=$(git diff --cached --name-only | wc -l) - - if [ "$STAGED_COUNT" -gt 0 ]; then - # Commit and push - git commit -m "docs: Stage 1 - Inline documentation ($STAGE_FILES files) [skip ci]" - git push origin "$BRANCH_NAME" - - echo "Committed and pushed $STAGED_COUNT stage 1 files" - set_output "committed" "true" - elif [ -s "$STAGE1_PUSH_MARKER" ]; then - # Everything already landed via the incremental progress pushes - echo "Nothing left to commit: every stage 1 file was pushed during generation" - set_output "committed" "true" - else - echo "::warning title=Stage 1 produced nothing::No inline docs to commit." - set_output "committed" "false" - fi - - # ========================================================================= - # ORPHANED INLINE DOCS - # Right after stage 1: removes each hidden .*.md whose source file is gone. - # ========================================================================= - # `== 'completed'` HERE IS DELIBERATE, unlike the commit steps: deleting - # "orphaned" docs after a stage that FAILED would delete docs whose - # sources were never re-examined. - - name: Remove orphaned inline docs - if: always() && steps.stage1.outputs.stage1_status == 'completed' - id: orphan-detection-inline - env: - BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }} - run: | - source /tmp/workflow-helpers.sh - - # Run orphan detection script - bash /tmp/detect-orphans.sh || true - - # Check if orphans were deleted (script writes list to /tmp/orphaned-inline-files.txt) - DELETED_FILES_LIST="/tmp/orphaned-inline-files.txt" - - if [ -f "$DELETED_FILES_LIST" ] && [ -s "$DELETED_FILES_LIST" ]; then - DELETED_COUNT=$(wc -l < "$DELETED_FILES_LIST" | tr -d ' ') - echo "Orphaned inline docs removed: $DELETED_COUNT" - - # Only add the specific files that were deleted by the script - while IFS= read -r deleted_file; do - git add "$deleted_file" 2>/dev/null || true - done < "$DELETED_FILES_LIST" - - # Verify we have staged changes - STAGED_COUNT=$(git diff --cached --name-only | wc -l | tr -d ' ') - - if [ "$STAGED_COUNT" -gt 0 ]; then - git commit -m "chore(docs): Remove $DELETED_COUNT orphaned inline files [skip ci]" - git push origin "$BRANCH_NAME" - echo "Committed $STAGED_COUNT orphan removals" - set_output "orphans_deleted" "$STAGED_COUNT" - else - echo "Nothing to commit (the files were already absent)" - set_output "orphans_deleted" "0" - fi - else - echo "No orphaned inline docs" - set_output "orphans_deleted" "0" - fi - - - name: Update pull request with stage 1 progress - if: always() && steps.commit-stage1.outputs.committed == 'true' && steps.create-initial-pr.outputs.pull-request-number - env: - GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} - PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }} - STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files }} - RUN_ID_VAR: ${{ env.RUN_ID }} - REPO_NAME: ${{ github.repository }} - run: | - source /tmp/workflow-helpers.sh - - # Update PR body - PR_BODY="## ๐Ÿฆฉ Flamingo Code Documentation: In Progress - - **Run ID:** \`$RUN_ID_VAR\` - **Status:** ๐Ÿ”„ Running... - - This PR is being updated as each documentation stage completes. - - ### Progress - - โœ… Stage 1: Inline Documentation - Completed ($STAGE1_FILES files) - - โณ Stage 2: Architecture Analysis - Running... - - โฑ๏ธ Stage 3: Tutorial Generation - Pending - - โฑ๏ธ Stage 4: Repository Documentation - Pending - - --- - ๐Ÿฆฉ Generated by [Flamingo Code Documentation](https://flamingo.run)" - - gh pr edit "$PR_NUMBER" --body "$PR_BODY" - - echo "Pull request #$PR_NUMBER updated with stage 1 progress" - - - name: Report stage 1 progress - if: always() && contains(env.STAGES, 'inline-docs') && env.HUB_BASE_URL != '' - continue-on-error: true - env: - WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }} - WORKFLOW_RUN_ID: ${{ github.run_id }} - WORKFLOW_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} - STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }} - STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }} - PR_URL: ${{ steps.create-initial-pr.outputs.pull-request-url }} - PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }} - run: | - source /tmp/workflow-helpers.sh - - CALLBACK_URL="${HUB_BASE_URL}/api/code-documentation/webhook" - echo "Reporting stage 1 (inline docs): $STAGE1_STATUS, $STAGE1_FILES files" - report_stage_progress "$CALLBACK_URL" "$WEBHOOK_SECRET" "$RUN_ID" "$REPO_ID" \ - "$WORKFLOW_RUN_ID" "$WORKFLOW_URL" "codewiki" \ - "$STAGE1_STATUS" "$STAGE1_FILES" "" "0" "" "0" "" "0" \ - "$PR_URL" "$PR_NUMBER" - - # ========================================================================= - # ECOSYSTEM FACTS โ€” derived by the hub from the code graph, never written - # by a model. Two renderings of the same live snapshot: ecosystem.md - # (committed under the reference tree and fed to the Stage 2/3/4 prompts - # as ground truth for the Dependencies sections) and the marker-delimited - # AGENTS.md block (upserted in place, idempotent). A repo with no graph - # yet is a notice, not a failure. - # ========================================================================= - - name: Fetch ecosystem facts - continue-on-error: true - env: - WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }} - REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }} - run: | - source /tmp/workflow-helpers.sh - - # Through ci-hub.mjs, the shell's way into the ONE hub transport - # (code-review-lib.mjs): it checks the destination before the secret - # leaves, keeps the secret in a header, and writes the body only on a - # 2xx. It prints the status and exits 0 whenever the hub ANSWERED โ€” a 404 - # is an answer ("no graph yet"), not a failure. - REPO_PARAM=$(printf '%s' "$GITHUB_REPOSITORY" | sed 's|/|%2F|g') - ECOSYSTEM_PATH="/api/ci/code-graph/ecosystem.md?repo=${REPO_PARAM}" - - HTTP_CODE=$(node /tmp/ci-hub.mjs get "$ECOSYSTEM_PATH" /tmp/ecosystem.md) || HTTP_CODE="000" - if [ "$HTTP_CODE" = "404" ]; then - echo "::notice title=No code graph yet::The hub has no code graph for $GITHUB_REPOSITORY yet; the docs are written without ecosystem facts." - rm -f /tmp/ecosystem.md - exit 0 - fi - if [ "$HTTP_CODE" != "200" ]; then - echo "::warning title=Ecosystem facts unavailable::The hub answered HTTP $HTTP_CODE; the Dependencies sections are written without cross-repository facts." - rm -f /tmp/ecosystem.md - exit 0 - fi - # Kept in /tmp ONLY until Stage 2 has run. The reference directory is - # CodeWiki's output directory, and CodeWiki asks "already contains - # documentation. Overwrite?" when it finds a .md file there โ€” a prompt - # a runner cannot answer, so Stage 2 aborted on every CodeWiki repo - # (CodeWiki run 35293807658). The Stage 2 commit step copies the file - # into place, after either engine has written its own output. - echo "Fetched ecosystem.md ($(wc -c < /tmp/ecosystem.md | tr -d ' ') bytes); it is copied to $REFERENCE_OUTPUT_PATH after stage 2" - - HTTP_CODE=$(node /tmp/ci-hub.mjs get "${ECOSYSTEM_PATH}&format=agents" /tmp/ecosystem-agents.md) || HTTP_CODE="000" - if [ "$HTTP_CODE" = "200" ]; then - upsert_marker_block AGENTS.md /tmp/ecosystem-agents.md - else - echo "::warning title=AGENTS.md block unavailable::The hub answered HTTP $HTTP_CODE; AGENTS.md is left untouched." - rm -f /tmp/ecosystem-agents.md - fi - - # ========================================================================= - # STAGE 2: REFERENCE DOCS - # Architecture overview, module tree and diagrams. CodeWiki where it can - # parse the primary language ("Detect primary language" decided, before - # stage 1), otherwise the Claude architecture analysis. - # ========================================================================= - - - name: Set up Python 3.12 - if: contains(env.STAGES, 'codewiki') && steps.detect_language.outputs.codewiki_supported == 'true' - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 - with: - python-version: '3.12' - - - name: Install CodeWiki - id: codewiki_install - if: contains(env.STAGES, 'codewiki') && steps.detect_language.outputs.codewiki_supported == 'true' - continue-on-error: true - env: - CODEWIKI_MAX_FILES_PER_MODULE: ${{ env.CODEWIKI_MAX_FILES_PER_MODULE }} - CODEWIKI_REPO: ${{ env.CODEWIKI_REPO }} - run: | - # Install keyrings.alt for headless keyring support in CI environments - # Install ipython to suppress "Mermaidjs magic function not available" warning - # Install colorama for CodeWiki colored terminal output - pip install keyrings.alt ipython colorama - - # Clone CodeWiki directly (no pip caching issues) - # Fixes baked into fork: - # - retries=3 for Pydantic AI agents (prevents "Tool exceeded max retries count of 1") - # - Synthetic module creation when clustering returns 0 modules (prevents context overflow) - # - 'children' key fix for synthetic modules - # - module_tree.json path fix (commit c1dfe5c) - loads from base docs dir, not nested module dir - # See: https://github.com/flamingo-stack/CodeWiki - # Extract repo URL from CODEWIKI_REPO (strip git+ prefix and @branch/commit suffix) - REPO_URL=$(echo "$CODEWIKI_REPO" | sed 's|^git+||' | sed 's|@[^@]*$||') - REF=$(echo "$CODEWIKI_REPO" | grep -o '@[^@]*$' | sed 's|^@||' || echo "main") - echo "๐Ÿ“ฆ Cloning CodeWiki from: $REPO_URL (ref: ${REF:-main})" - rm -rf /tmp/CodeWiki - - # Clone and checkout - handle both branches and commit hashes - if [[ "${REF}" =~ ^[0-9a-f]{7,40}$ ]]; then - # Commit hash - clone full repo and checkout specific commit - git clone "$REPO_URL" /tmp/CodeWiki - cd /tmp/CodeWiki && git checkout "${REF}" && cd - - else - # Branch name - shallow clone - git clone --depth 1 --branch "${REF:-main}" "$REPO_URL" /tmp/CodeWiki - fi - - echo " Commit: $(cd /tmp/CodeWiki && git rev-parse --short HEAD)" - - # Install from local clone (reliable, no caching) - echo "๐Ÿ“ฆ Installing CodeWiki from local clone..." - pip install --no-cache-dir /tmp/CodeWiki - - source /tmp/workflow-helpers.sh - set_output "codewiki_installed" "true" - - - name: Configure CodeWiki - id: codewiki_config - if: contains(env.STAGES, 'codewiki') && steps.detect_language.outputs.codewiki_supported == 'true' && steps.codewiki_install.outputs.codewiki_installed == 'true' - continue-on-error: true - env: - # SECURITY: Pass secrets per-step with inline masking - OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} - ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} - # Set keyring backend via env var (must be set before any keyring operations) - PYTHON_KEYRING_BACKEND: keyrings.alt.file.PlaintextKeyring - # Flamingo Markdown Guidelines path (needed for module import during config/validate) - FLAMINGO_MARKDOWN_GUIDELINES_PATH: /tmp/flamingo-markdown-guidelines.md - # OSS Tenant Structure: Stage 2 outputs (for clean slate deletion) - REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }} - DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }} - run: | - source /tmp/workflow-helpers.sh - - # === KEYRING CONFIGURATION FOR CI === - # CodeWiki stores API keys in system keyring. In CI (no GUI), we must: - # 1. Create keyring config to specify PlaintextKeyring backend - # 2. Create data directory for credential storage - # See: https://github.com/FSoft-AI4Code/CodeWiki - uses keyring.set_password() - - echo "๐Ÿ”‘ Setting up keyring for headless CI environment..." - - # Create keyring configuration directory and config file - mkdir -p ~/.config/python_keyring - cat > ~/.config/python_keyring/keyringrc.cfg << 'KEYRING_CFG' - [backend] - default-keyring=keyrings.alt.file.PlaintextKeyring - KEYRING_CFG - - # Ensure keyring data directory exists with proper permissions - mkdir -p ~/.local/share/python_keyring - chmod 700 ~/.local/share/python_keyring - - # Debug: Verify keyring is properly configured - echo "๐Ÿ“‹ Keyring backend verification:" - python3 -c "import keyring; print(f' Active backend: {keyring.get_keyring()}')" - - # Configure CodeWiki with separate cluster and generation providers/models - # CodeWiki calls provider APIs directly via --base-url - # Model names should match the provider's API format (no LiteLLM prefix needed) - # OpenAI: gpt-4o, gpt-4-turbo, gpt-4o-mini - # Provider ids come from MODEL_METADATA in lib/constants/ai-models.ts - # See: https://github.com/FSoft-AI4Code/CodeWiki - - echo "๐Ÿ”ง Configuring CodeWiki..." - echo " Cluster (Phase 2): $CODEWIKI_CLUSTER_PROVIDER / $CODEWIKI_CLUSTER_MODEL" - echo " Generation (Phase 3+): $CODEWIKI_GENERATION_PROVIDER / $CODEWIKI_GENERATION_MODEL" - - # Source helper functions for configuration - source /tmp/workflow-helpers.sh - - # Determine API keys for each provider (cluster, generation/main, fallback) - # Each provider can use a different AI service (OpenAI, Anthropic, etc.) - if [ "$CODEWIKI_CLUSTER_PROVIDER" = "anthropic" ]; then - CLUSTER_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}" - echo " Cluster: Using Anthropic API key" - else - CLUSTER_API_KEY="${{ secrets.OPENAI_API_KEY }}" - echo " Cluster: Using OpenAI API key" - fi - - if [ "$CODEWIKI_GENERATION_PROVIDER" = "anthropic" ]; then - MAIN_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}" - echo " Generation: Using Anthropic API key" - else - MAIN_API_KEY="${{ secrets.OPENAI_API_KEY }}" - echo " Generation: Using OpenAI API key" - fi - - if [ "$CODEWIKI_FALLBACK_PROVIDER" = "anthropic" ]; then - FALLBACK_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}" - echo " Fallback: Using Anthropic API key" - else - FALLBACK_API_KEY="${{ secrets.OPENAI_API_KEY }}" - echo " Fallback: Using OpenAI API key" - fi - - # Configure CodeWiki from CODEWIKI_CONFIG_JSON (single extractor: configure_codewiki_from_json) - # Pass per-provider API keys for mixed provider configurations - configure_codewiki_from_json "$CODEWIKI_CONFIG_JSON" "$CLUSTER_API_KEY" "$MAIN_API_KEY" "$FALLBACK_API_KEY" - - if [ $? -ne 0 ]; then - echo "โŒ CodeWiki configuration failed" - exit 1 - fi - - # Set environment variables for backward compatibility with run-codewiki-analysis.sh - export MAIN_MODEL="$CODEWIKI_GENERATION_MODEL" - export FALLBACK_MODEL_1="$CODEWIKI_FALLBACK_MODEL" - if [ "$PRIMARY_PROVIDER" = "anthropic" ]; then - export ANTHROPIC_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}" - else - export OPENAI_API_KEY="${{ secrets.OPENAI_API_KEY }}" - fi - - # Verify configuration was saved - echo "" - echo "๐Ÿ“‹ CodeWiki configuration:" - python -m codewiki config show - - echo "" - echo "โœ… Validating configuration..." - python -m codewiki config validate - - echo "" - set_output "codewiki_configured" "true" - - - name: Generate reference docs with CodeWiki (stage 2) - id: stage2 - if: contains(env.STAGES, 'codewiki') && steps.detect_language.outputs.codewiki_supported == 'true' && steps.codewiki_config.outputs.codewiki_configured == 'true' - continue-on-error: false - env: - # Keyring backend for CI (must match config step) - PYTHON_KEYRING_BACKEND: keyrings.alt.file.PlaintextKeyring - # API keys for both providers (CodeWiki will use the one configured) - OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} - ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} - # OSS Tenant Structure: Stage 2 outputs - REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }} - DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }} - # Stage timeout - STAGE2_TIMEOUT_HOURS: ${{ env.STAGE2_TIMEOUT_HOURS }} - # CodeWiki JSON configuration (required for unified function) - CODEWIKI_CONFIG_JSON: ${{ env.CODEWIKI_CONFIG_JSON }} - # CodeWiki model configuration (cluster, generation, fallback) - CODEWIKI_CLUSTER_PROVIDER: ${{ env.CODEWIKI_CLUSTER_PROVIDER }} - CODEWIKI_CLUSTER_MODEL: ${{ env.CODEWIKI_CLUSTER_MODEL }} - CODEWIKI_CLUSTER_MAX_TOKENS: ${{ env.CODEWIKI_CLUSTER_MAX_TOKENS }} - CODEWIKI_CLUSTER_MAX_TOKEN_FIELD: ${{ env.CODEWIKI_CLUSTER_MAX_TOKEN_FIELD }} - CODEWIKI_GENERATION_PROVIDER: ${{ env.CODEWIKI_GENERATION_PROVIDER }} - CODEWIKI_GENERATION_MODEL: ${{ env.CODEWIKI_GENERATION_MODEL }} - CODEWIKI_GENERATION_MAX_TOKENS: ${{ env.CODEWIKI_GENERATION_MAX_TOKENS }} - CODEWIKI_GENERATION_MAX_TOKEN_FIELD: ${{ env.CODEWIKI_GENERATION_MAX_TOKEN_FIELD }} - CODEWIKI_FALLBACK_PROVIDER: ${{ env.CODEWIKI_FALLBACK_PROVIDER }} - CODEWIKI_FALLBACK_MODEL: ${{ env.CODEWIKI_FALLBACK_MODEL }} - CODEWIKI_FALLBACK_MAX_TOKENS: ${{ env.CODEWIKI_FALLBACK_MAX_TOKENS }} - CODEWIKI_FALLBACK_MAX_TOKEN_FIELD: ${{ env.CODEWIKI_FALLBACK_MAX_TOKEN_FIELD }} - CODEWIKI_MAX_DEPTH: ${{ env.CODEWIKI_MAX_DEPTH }} - # Flamingo Markdown Guidelines path for CodeWiki prompts - FLAMINGO_MARKDOWN_GUIDELINES_PATH: /tmp/flamingo-markdown-guidelines.md - # Markdown Validation Rules (injected into all prompts) - VALIDATION_RULES_PATH: ${{ env.VALIDATION_RULES_PATH }} - # Custom AI Instructions (All Stages) - CUSTOM_REPO_INSTRUCTIONS: ${{ env.CUSTOM_INSTRUCTIONS }} - # External Repositories - EXTERNAL_REPOS: ${{ env.EXTERNAL_REPOS }} - # Dependencies (for CodeWiki multi-path support) - DEPENDENCIES: ${{ env.DEPENDENCIES }} - run: | - # Determine per-provider API keys (same logic as Configure step) - if [ "$CODEWIKI_CLUSTER_PROVIDER" = "anthropic" ]; then - export CLUSTER_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}" - else - export CLUSTER_API_KEY="${{ secrets.OPENAI_API_KEY }}" - fi - - if [ "$CODEWIKI_GENERATION_PROVIDER" = "anthropic" ]; then - export MAIN_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}" - else - export MAIN_API_KEY="${{ secrets.OPENAI_API_KEY }}" - fi - - if [ "$CODEWIKI_FALLBACK_PROVIDER" = "anthropic" ]; then - export FALLBACK_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}" - else - export FALLBACK_API_KEY="${{ secrets.OPENAI_API_KEY }}" - fi - - # Verify dependencies directory before CodeWiki runs - echo "" - echo "๐Ÿ” Pre-CodeWiki Dependency Verification:" - echo " Current directory: $(pwd)" - echo " Absolute path: $(realpath .)" - echo "" - - if [ -d "./deps" ]; then - echo " โœ… ./deps EXISTS" - echo " Contents: $(ls -1 ./deps 2>/dev/null | wc -l) repositories" - ls -la ./deps 2>/dev/null | head -5 - else - echo " โŒ ./deps NOT FOUND" - fi - - if [ -d "../deps" ]; then - echo " โœ… ../deps EXISTS" - echo " Absolute: $(realpath ../deps)" - echo " Contents: $(ls -1 ../deps 2>/dev/null | wc -l) repositories" - ls -la ../deps 2>/dev/null | head -5 - else - echo " โŒ ../deps NOT FOUND" - fi - - echo " DEPENDENCIES env: ${DEPENDENCIES:-}" - echo "" - - # Run externalized CodeWiki analysis script - /tmp/run-codewiki-analysis.sh - - # The alternative for any language CodeWiki cannot parse. - - name: Generate reference docs with Claude (stage 2) - id: stage2_alt - if: contains(env.STAGES, 'codewiki') && steps.detect_language.outputs.codewiki_supported == 'false' - env: - # No ANTHROPIC_API_KEY โ€” this stage calls Claude through the hub - # (/api/ci/claude), which the WEBHOOK_SECRET below authenticates. - WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }} - # SSOT โ€” see workflow env CLAUDE_MODEL block - CLAUDE_MODEL: ${{ env.CLAUDE_MODEL }} - PRIMARY_LANGUAGE: ${{ steps.detect_language.outputs.primary_language }} - # OSS Tenant Structure: Stage 2 outputs - REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }} - DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }} - # Unified file discovery result (single source of truth) - SOURCE_FILES_LIST: ${{ steps.discover_files.outputs.source_files_list }} - SOURCE_FILE_COUNT: ${{ steps.discover_files.outputs.source_file_count }} - # Custom AI Instructions (All Stages) - CUSTOM_REPO_INSTRUCTIONS: ${{ env.CUSTOM_INSTRUCTIONS }} - # External Repositories - EXTERNAL_REPOS: ${{ env.EXTERNAL_REPOS }} - # Dependencies (for consistency with other stages) - DEPENDENCIES: ${{ env.DEPENDENCIES }} - run: | - # Run externalized Claude architecture analysis script - /tmp/run-claude-architecture-analysis.sh - - # ========================================================================= - # RESTORE DOCS NO STAGE REGENERATED - # The docs removal took the reference/diagrams trees because `codewiki` was in - # STAGES. If neither Stage-2 variant then completed โ€” a repo with no - # discoverable source, an install failure, a skip โ€” the deletion would be the - # only Stage-2 change in the pull request, i.e. a net removal of documentation - # nobody asked to remove. Put it back. - # ========================================================================= - - name: Restore docs no stage regenerated - if: always() - env: - BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }} - STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status }} - STAGE2_STATUS: ${{ steps.stage2.outputs.stage2_status }} - STAGE2_ALT_STATUS: ${{ steps.stage2_alt.outputs.stage2_status }} - run: | - # The docs removal deleted whatever the configured stages own, and committed that - # deletion. Any owned subtree whose stage then produced nothing must be put - # back โ€” otherwise the pull request is a net REMOVAL of documentation nobody - # asked to remove. Iterates the exact list the removal recorded, so the - # stage -> subtree map has one home. - if [ -z "$CLEAN_TARGETS_RECORD" ]; then - echo "Nothing was removed; nothing to restore" - exit 0 - fi - - STAGE2_OK=false - [ "$STAGE2_STATUS" = "completed" ] && STAGE2_OK=true - [ "$STAGE2_ALT_STATUS" = "completed" ] && STAGE2_OK=true - - # CodeWiki leaves multi-GB scratch behind on a failed run; never let it near - # the index. - rm -rf "$REFERENCE_OUTPUT_PATH/temp" "$DIAGRAMS_OUTPUT_PATH/temp" 2>/dev/null || true - - RESTORED=0 - while IFS= read -r target; do - [ -n "$target" ] || continue - # A target is safe to leave deleted only if something regenerated it. - case "$target" in - "$REFERENCE_OUTPUT_PATH"|"$DIAGRAMS_OUTPUT_PATH") - [ "$STAGE2_OK" = true ] && continue ;; - "$GETTING_STARTED_OUTPUT_PATH"|"$DEVELOPMENT_OUTPUT_PATH") - [ "$STAGE3_STATUS" = "completed" ] && continue ;; - "$DOCS_OUTPUT_PATH") - # Full wipe: only fully safe when every stage delivered. - if [ "$STAGE1_STATUS" = "completed" ] && [ "$STAGE2_OK" = true ] && \ - [ "$STAGE3_STATUS" = "completed" ] && [ "$STAGE4_STATUS" = "completed" ]; then - continue - fi ;; - esac - if git checkout "$PRE_CLEAN_SHA" -- "$target" 2>/dev/null; then - echo "::notice title=Docs restored::$target restored from $PRE_CLEAN_SHA; no stage regenerated it." - RESTORED=1 - fi - done <<< "$CLEAN_TARGETS_RECORD" - - # `git checkout -- ` already stages exactly those paths. Deliberately NO - # `git add -A`: at this point the workspace holds npm install output from - # Stage 1 and, on a failed CodeWiki run, its scratch trees. - if [ "$RESTORED" -eq 1 ] && [ -n "$(git diff --cached --name-only)" ]; then - git commit -m "chore(docs): restore documentation no stage regenerated [skip ci]" - git push origin "$BRANCH_NAME" - echo "Restore committed and pushed" - else - echo "Nothing to restore" - fi - - # ========================================================================= - # COMMIT STAGE 2 RESULTS (progressive pull request) - # ========================================================================= - - name: Commit and push stage 2 results - if: always() && (steps.stage2.outputs.stage2_status == 'completed' || steps.stage2_alt.outputs.stage2_status == 'completed') - id: commit-stage2 - env: - BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }} - STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files }} - REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }} - DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }} - run: | - source /tmp/workflow-helpers.sh - - # Remove CodeWiki scratch from every output directory - rm -rf "$REFERENCE_OUTPUT_PATH/temp" 2>/dev/null || echo "::warning::Could not remove $REFERENCE_OUTPUT_PATH/temp" - rm -rf "$DIAGRAMS_OUTPUT_PATH/temp" 2>/dev/null || echo "::warning::Could not remove $DIAGRAMS_OUTPUT_PATH/temp" - - # Verify cleanup - if [ -d "$REFERENCE_OUTPUT_PATH/temp" ]; then - echo "::error title=Scratch not removed::$REFERENCE_OUTPUT_PATH/temp is still present; refusing to commit it." - ls -la "$REFERENCE_OUTPUT_PATH/temp" - exit 1 - fi - - # .gitignore in each output directory keeps CodeWiki scratch out of commits - for output_dir in "$REFERENCE_OUTPUT_PATH" "$DIAGRAMS_OUTPUT_PATH"; do - if [ -d "$output_dir" ]; then - { - echo "# CodeWiki temp files (dependency graphs can be 7GB+)" - echo "temp/" - echo "dependency_graphs/" - echo "" - echo "# JSON intermediate files (except schema/config)" - echo "*.json" - echo "!*-schema.json" - echo "!*-config.json" - } > "$output_dir/.gitignore" - echo "Wrote $output_dir/.gitignore" - fi - done - - echo "::group::Stage 2 output before staging" - echo "Reference directory ($REFERENCE_OUTPUT_PATH):" - if [ -d "$REFERENCE_OUTPUT_PATH" ]; then - find "$REFERENCE_OUTPUT_PATH" -type f \( -name "*.md" -o -name "*.mmd" -o -name ".gitignore" \) | head -20 - FILE_COUNT=$(find "$REFERENCE_OUTPUT_PATH" -type f \( -name "*.md" -o -name "*.mmd" \) | wc -l | tr -d ' ') - echo ".md/.mmd files: $FILE_COUNT" - else - echo "(directory does not exist)" - fi - echo "Diagrams directory ($DIAGRAMS_OUTPUT_PATH):" - if [ -d "$DIAGRAMS_OUTPUT_PATH" ]; then - find "$DIAGRAMS_OUTPUT_PATH" -type f \( -name "*.md" -o -name "*.mmd" -o -name ".gitignore" \) | head -20 - FILE_COUNT=$(find "$DIAGRAMS_OUTPUT_PATH" -type f \( -name "*.md" -o -name "*.mmd" \) | wc -l | tr -d ' ') - echo ".md/.mmd files: $FILE_COUNT" - else - echo "(directory does not exist)" - fi - echo "::endgroup::" - - # The hub-rendered ecosystem.md joins the reference directory only NOW, - # after the engine has run: placed earlier it makes CodeWiki prompt for - # an overwrite (see "Fetch ecosystem facts"). - if [ -f /tmp/ecosystem.md ]; then - mkdir -p "$REFERENCE_OUTPUT_PATH" - cp /tmp/ecosystem.md "$REFERENCE_OUTPUT_PATH/ecosystem.md" - echo "Added ecosystem.md to $REFERENCE_OUTPUT_PATH/" - fi - - # Stage all .md, .mmd, and .gitignore files from Stage 2 output directories - # CRITICAL: Use git add on full paths to preserve nested directory structure - # This ensures Backend/Authentication/JWT/JWT.md keeps its full path in git - - # Add only .md, .mmd, .gitignore, and allowed JSON files (*-schema.json, *-config.json) - # This excludes CodeWiki intermediate files: module_tree.json, first_module_tree.json, metadata.json - if [ -d "$REFERENCE_OUTPUT_PATH" ]; then - find "$REFERENCE_OUTPUT_PATH" -type f \( \ - -name "*.md" -o -name "*.mmd" -o -name ".gitignore" \ - -o -name "*-schema.json" -o -name "*-config.json" \ - \) -exec git add {} \; 2>/dev/null || true - echo "Staged .md/.mmd/.gitignore/*-schema.json/*-config.json under $REFERENCE_OUTPUT_PATH/" - fi - - if [ -d "$DIAGRAMS_OUTPUT_PATH" ]; then - find "$DIAGRAMS_OUTPUT_PATH" -type f \( \ - -name "*.md" -o -name "*.mmd" -o -name ".gitignore" \ - -o -name "*-schema.json" -o -name "*-config.json" \ - \) -exec git add {} \; 2>/dev/null || true - echo "Staged .md/.mmd/.gitignore/*-schema.json/*-config.json under $DIAGRAMS_OUTPUT_PATH/" - fi - - # AGENTS.md carries the hub-rendered ecosystem block ("Fetch ecosystem - # facts" upserts it just before this stage). It sits at the repository - # root, outside every output directory staged above, so it is staged by - # name: the first production run wrote the block and no commit ever - # picked the file up (openframe-cli#383). - if [ -f AGENTS.md ]; then - git add -f AGENTS.md - echo "Staged AGENTS.md (ecosystem block)" - fi - - # Check if there are changes - STAGED_COUNT=$(git diff --cached --name-only | wc -l) - echo "Staged files: $STAGED_COUNT" - - if [ "$STAGED_COUNT" -gt 0 ]; then - echo "::group::Staged stage 2 files (first 30)" - git diff --cached --name-only | head -30 - echo "::endgroup::" - fi - - if [ "$STAGED_COUNT" -gt 0 ]; then - # Commit and push - git commit -m "docs: Stage 2 - Architecture analysis ($STAGE2_FILES files) [skip ci]" - git push origin "$BRANCH_NAME" - - echo "Committed and pushed $STAGED_COUNT stage 2 files" - set_output "committed" "true" - else - echo "::error title=Stage 2 output not staged::Stage 2 reported $STAGE2_FILES files but none were staged; the files were generated outside the reference and diagrams directories, or not at all." - set_output "committed" "false" - fi - - - name: Update pull request with stage 2 progress - if: always() && steps.commit-stage2.outputs.committed == 'true' && steps.create-initial-pr.outputs.pull-request-number - env: - GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} - PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }} - STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }} - STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }} - RUN_ID_VAR: ${{ env.RUN_ID }} - REPO_NAME: ${{ github.repository }} - run: | - source /tmp/workflow-helpers.sh - - # Update PR body - PR_BODY="## ๐Ÿฆฉ Flamingo Code Documentation: In Progress - - **Run ID:** \`$RUN_ID_VAR\` - **Status:** ๐Ÿ”„ Running... - - This PR is being updated as each documentation stage completes. - - ### Progress - - โœ… Stage 1: Inline Documentation - Completed ($STAGE1_FILES files) - - โœ… Stage 2: Architecture Analysis - Completed ($STAGE2_FILES files) - - โณ Stage 3: Tutorial Generation - Running... - - โฑ๏ธ Stage 4: Repository Documentation - Pending - - --- - ๐Ÿฆฉ Generated by [Flamingo Code Documentation](https://flamingo.run)" - - gh pr edit "$PR_NUMBER" --body "$PR_BODY" - - echo "Pull request #$PR_NUMBER updated with stage 2 progress" - - - name: Report stage 2 progress - if: always() && contains(env.STAGES, 'codewiki') && env.HUB_BASE_URL != '' - continue-on-error: true - env: - WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }} - WORKFLOW_RUN_ID: ${{ github.run_id }} - WORKFLOW_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} - STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }} - STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }} - # Use outputs from either CodeWiki (stage2) or Claude alternative (stage2_alt) - STAGE2_STATUS: ${{ steps.stage2.outputs.stage2_status || steps.stage2_alt.outputs.stage2_status || 'skipped' }} - STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }} - PR_URL: ${{ steps.create-initial-pr.outputs.pull-request-url }} - PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }} - run: | - source /tmp/workflow-helpers.sh - - CALLBACK_URL="${HUB_BASE_URL}/api/code-documentation/webhook" - echo "Reporting stage 2 (reference docs): $STAGE2_STATUS, $STAGE2_FILES files" - report_stage_progress "$CALLBACK_URL" "$WEBHOOK_SECRET" "$RUN_ID" "$REPO_ID" \ - "$WORKFLOW_RUN_ID" "$WORKFLOW_URL" "tutorials" \ - "$STAGE1_STATUS" "$STAGE1_FILES" "$STAGE2_STATUS" "$STAGE2_FILES" "" "0" "" "0" \ - "$PR_URL" "$PR_NUMBER" - - # ========================================================================= - # STAGE 3: TUTORIALS - # Getting-started guides and how-to tutorials, written by the Code - # Documentation lane (code-documentation-lib.mjs): one hub call per - # tutorial with the forced `emit_document` tool and, when the repository's - # "Graph lookups" switch is on, the hub's read tools (the code graph and - # the rules, scoped to what THIS repository may see). No model key here. - # Generates 4 tutorials: user/getting-started, user/common-use-cases, - # dev/getting-started-dev, dev/architecture-overview-dev - # ========================================================================= - - name: Install tutorial and repository docs dependencies - # Stage 3 AND Stage 4 (generate-repo-docs.cjs) share this tree. Gated - # on either: a repository configured with inline-docs + repo-docs and no - # tutorials reached Stage 4 with no dependencies at all, run 35302244804. - if: contains(env.STAGES, 'tutorials') || contains(env.STAGES, 'repo-docs') - # Isolated deps tree (see Stage 1) - no reconciliation with the target repo. - # No model SDK: both stages call Claude through the hub. - run: | - mkdir -p "$RUNNER_TEMP/doc-orch-deps" && cd "$RUNNER_TEMP/doc-orch-deps" - npm init -y >/dev/null 2>&1 - npm install zod@3.25.76 glob@13.0.6 - - - name: Generate tutorials (stage 3) - id: stage3 - if: contains(env.STAGES, 'tutorials') - env: - # No ANTHROPIC_API_KEY โ€” this stage calls Claude through the hub - # (/api/ci/claude), which the WEBHOOK_SECRET below authenticates. - WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }} - # Pass through output paths from workflow env (OSS Tenant Structure) - DOCS_OUTPUT_PATH: ${{ env.DOCS_OUTPUT_PATH }} - # Stage 2 outputs (for context) - REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }} - DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }} - # Stage 3 outputs - GETTING_STARTED_OUTPUT_PATH: ${{ env.GETTING_STARTED_OUTPUT_PATH }} - DEVELOPMENT_OUTPUT_PATH: ${{ env.DEVELOPMENT_OUTPUT_PATH }} - # Claude model SSOT โ€” see workflow env CLAUDE_MODEL block - CLAUDE_MODEL: ${{ env.CLAUDE_MODEL }} - # Stage timeout - STAGE3_TIMEOUT_HOURS: ${{ env.STAGE3_TIMEOUT_HOURS }} - # Unified file discovery result (same files as Stage 1 and 2) - SOURCE_FILES_LIST: ${{ steps.discover_files.outputs.source_files_list }} - # NODE_PATH to find modules from /tmp/ scripts - NODE_PATH: ${{ runner.temp }}/doc-orch-deps/node_modules - # YouTube Integration - SECURITY: API key from secrets, NOT dispatch payload - YOUTUBE_ENABLED: ${{ env.YOUTUBE_ENABLED }} - YOUTUBE_CHANNELS: ${{ env.YOUTUBE_CHANNELS }} - YOUTUBE_API_KEY: ${{ secrets.YOUTUBE_API_KEY }} - # Markdown Validation Rules (injected into prompts) - VALIDATION_RULES_PATH: ${{ env.VALIDATION_RULES_PATH }} - # Flamingo Markdown Guidelines (optional) - GUIDELINES_PATH: ${{ env.GUIDELINES_PATH }} - # Stage 3 tracking files (configurable paths) - STAGE3_FILES_TRACKER: ${{ env.STAGE3_FILES_TRACKER }} - STAGE3_STATS_FILE: ${{ env.STAGE3_STATS_FILE }} - # Custom AI Instructions (All Stages) - CUSTOM_REPO_INSTRUCTIONS: ${{ env.CUSTOM_INSTRUCTIONS }} - # Analysis Exclusions - EXCLUDED_PATHS: ${{ env.EXCLUDED_PATHS }} - # External Repositories - EXTERNAL_REPOS: ${{ env.EXTERNAL_REPOS }} - run: | - source /tmp/workflow-helpers.sh - - # Script already downloaded to /tmp/ in setup step. run_stage records - # the outcome as stage3_status โ€” see its note in workflow-helpers.sh. - run_stage "Stage 3" "$STAGE3_TIMEOUT_HOURS" stage3_status node /tmp/generate-tutorials-voltagent.cjs - - # Count files from both OSS Tenant Structure directories - GETTING_STARTED_FILES=$(count_markdown_files "${GETTING_STARTED_OUTPUT_PATH}") - DEVELOPMENT_FILES=$(count_markdown_files "${DEVELOPMENT_OUTPUT_PATH}") - TUTORIAL_FILES=$((GETTING_STARTED_FILES + DEVELOPMENT_FILES)) - echo "Tutorials written: $TUTORIAL_FILES ($GETTING_STARTED_FILES getting started, $DEVELOPMENT_FILES development)" - set_output "stage3_files" "$TUTORIAL_FILES" - - # ========================================================================= - # COMMIT STAGE 3 RESULTS (progressive pull request) - # ========================================================================= - # `!= ''` โ€” the run_stage commit rule, stated once at Stage 1. - - name: Commit and push stage 3 results - if: always() && steps.stage3.outputs.stage3_status != '' - id: commit-stage3 - env: - BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }} - STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files }} - GETTING_STARTED_OUTPUT_PATH: ${{ env.GETTING_STARTED_OUTPUT_PATH }} - DEVELOPMENT_OUTPUT_PATH: ${{ env.DEVELOPMENT_OUTPUT_PATH }} - run: | - source /tmp/workflow-helpers.sh - - # .gitignore in each tutorial output directory - for output_dir in "$GETTING_STARTED_OUTPUT_PATH" "$DEVELOPMENT_OUTPUT_PATH"; do - if [ -d "$output_dir" ]; then - { - echo "# VoltAgent temp files" - echo "temp/" - echo "" - echo "# JSON intermediate files (except schema/config)" - echo "*.json" - echo "!*-schema.json" - echo "!*-config.json" - } > "$output_dir/.gitignore" - echo "Wrote $output_dir/.gitignore" - fi - done - - # Stage all .md and .gitignore files from Stage 3 output directories - find "$GETTING_STARTED_OUTPUT_PATH" "$DEVELOPMENT_OUTPUT_PATH" -type f \( -name "*.md" -o -name ".gitignore" \) \ - -exec git add -f {} \; 2>/dev/null || true - - # Check if there are changes - STAGED_COUNT=$(git diff --cached --name-only | wc -l) - - if [ "$STAGED_COUNT" -gt 0 ]; then - # Commit and push - git commit -m "docs: Stage 3 - Tutorial generation ($STAGE3_FILES files) [skip ci]" - git push origin "$BRANCH_NAME" - - echo "Committed and pushed $STAGED_COUNT stage 3 files" - set_output "committed" "true" - else - echo "::warning title=Stage 3 produced nothing::No tutorials to commit." - set_output "committed" "false" - fi - - - name: Update pull request with stage 3 progress - if: always() && steps.commit-stage3.outputs.committed == 'true' && steps.create-initial-pr.outputs.pull-request-number - env: - GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} - PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }} - STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }} - STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }} - STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || 0 }} - RUN_ID_VAR: ${{ env.RUN_ID }} - REPO_NAME: ${{ github.repository }} - run: | - source /tmp/workflow-helpers.sh - - # Update PR body - PR_BODY="## ๐Ÿฆฉ Flamingo Code Documentation: In Progress - - **Run ID:** \`$RUN_ID_VAR\` - **Status:** ๐Ÿ”„ Running... - - This PR is being updated as each documentation stage completes. - - ### Progress - - โœ… Stage 1: Inline Documentation - Completed ($STAGE1_FILES files) - - โœ… Stage 2: Architecture Analysis - Completed ($STAGE2_FILES files) - - โœ… Stage 3: Tutorial Generation - Completed ($STAGE3_FILES files) - - โณ Stage 4: Repository Documentation - Running... - - --- - ๐Ÿฆฉ Generated by [Flamingo Code Documentation](https://flamingo.run)" - - gh pr edit "$PR_NUMBER" --body "$PR_BODY" - - echo "Pull request #$PR_NUMBER updated with stage 3 progress" - - - name: Report stage 3 progress - if: always() && contains(env.STAGES, 'tutorials') && env.HUB_BASE_URL != '' - continue-on-error: true - env: - WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }} - WORKFLOW_RUN_ID: ${{ github.run_id }} - WORKFLOW_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} - STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }} - STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }} - STAGE2_STATUS: ${{ steps.stage2.outputs.stage2_status || steps.stage2_alt.outputs.stage2_status || 'skipped' }} - STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }} - STAGE3_STATUS: ${{ steps.stage3.outputs.stage3_status || 'skipped' }} - STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || 0 }} - PR_URL: ${{ steps.create-initial-pr.outputs.pull-request-url }} - PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }} - run: | - source /tmp/workflow-helpers.sh - - CALLBACK_URL="${HUB_BASE_URL}/api/code-documentation/webhook" - echo "Reporting stage 3 (tutorials): $STAGE3_STATUS, $STAGE3_FILES files" - report_stage_progress "$CALLBACK_URL" "$WEBHOOK_SECRET" "$RUN_ID" "$REPO_ID" \ - "$WORKFLOW_RUN_ID" "$WORKFLOW_URL" "repo-docs" \ - "$STAGE1_STATUS" "$STAGE1_FILES" "$STAGE2_STATUS" "$STAGE2_FILES" "$STAGE3_STATUS" "$STAGE3_FILES" "" "0" \ - "$PR_URL" "$PR_NUMBER" - - # ========================================================================= - # STAGE 4: REPOSITORY DOCS - # Copies LICENSE.md, SECURITY.md from template repo - # Generates/updates README.md, CONTRIBUTING.md and the docs index through - # the Code Documentation lane (one hub call per document, no model key) - # ========================================================================= - - name: Generate repository docs (stage 4) - id: stage4 - if: contains(env.STAGES, 'repo-docs') - env: - # No ANTHROPIC_API_KEY โ€” this stage calls Claude through the hub. - WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }} - TEMPLATE_REPO: ${{ env.TEMPLATE_REPO }} - TEMPLATE_BRANCH: ${{ env.TEMPLATE_BRANCH }} - DOCS_OUTPUT_PATH: ${{ env.DOCS_OUTPUT_PATH }} - # OSS Tenant Structure: All output paths for docs/README.md navigation - REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }} - DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }} - GETTING_STARTED_OUTPUT_PATH: ${{ env.GETTING_STARTED_OUTPUT_PATH }} - DEVELOPMENT_OUTPUT_PATH: ${{ env.DEVELOPMENT_OUTPUT_PATH }} - # Claude model SSOT โ€” see workflow env CLAUDE_MODEL block - CLAUDE_MODEL: ${{ env.CLAUDE_MODEL }} - STAGE4_TIMEOUT_HOURS: ${{ env.STAGE4_TIMEOUT_HOURS }} - # NODE_PATH to find modules from /tmp/ scripts - NODE_PATH: ${{ runner.temp }}/doc-orch-deps/node_modules - # YouTube Integration - SECURITY: API key from secrets, NOT dispatch payload - YOUTUBE_ENABLED: ${{ env.YOUTUBE_ENABLED }} - YOUTUBE_CHANNELS: ${{ env.YOUTUBE_CHANNELS }} - YOUTUBE_API_KEY: ${{ secrets.YOUTUBE_API_KEY }} - # Markdown Validation Rules (injected into prompts) - VALIDATION_RULES_PATH: ${{ env.VALIDATION_RULES_PATH }} - # Flamingo Markdown Guidelines (optional) - GUIDELINES_PATH: ${{ env.GUIDELINES_PATH }} - # Stage 4 tracking files (configurable paths) - STAGE4_FILES_TRACKER: ${{ env.STAGE4_FILES_TRACKER }} - # Custom AI Instructions (All Stages) - CUSTOM_REPO_INSTRUCTIONS: ${{ env.CUSTOM_INSTRUCTIONS }} - # Analysis Exclusions - EXCLUDED_PATHS: ${{ env.EXCLUDED_PATHS }} - # External Repositories - EXTERNAL_REPOS: ${{ env.EXTERNAL_REPOS }} - # README Branding - README_LOGO_DARK: ${{ env.README_LOGO_DARK }} - README_LOGO_LIGHT: ${{ env.README_LOGO_LIGHT }} - README_LOGO_ALT: ${{ env.README_LOGO_ALT }} - run: | - source /tmp/workflow-helpers.sh - - echo "Template repository: $TEMPLATE_REPO@$TEMPLATE_BRANCH" - - RAW_URL="https://raw.githubusercontent.com/$TEMPLATE_REPO/$TEMPLATE_BRANCH" - - # 1. LICENSE.md and SECURITY.md from the template repository - if curl -fsSL "$RAW_URL/LICENSE.md" -o LICENSE.md 2>/dev/null; then - echo "Copied LICENSE.md from the template repository" - else - echo "::notice title=LICENSE.md not copied::The template repository $TEMPLATE_REPO@$TEMPLATE_BRANCH has no LICENSE.md." - fi - - if curl -fsSL "$RAW_URL/SECURITY.md" -o SECURITY.md 2>/dev/null; then - echo "Copied SECURITY.md from the template repository" - else - echo "::notice title=SECURITY.md not copied::The template repository $TEMPLATE_REPO@$TEMPLATE_BRANCH has no SECURITY.md." - fi - - # 2. The existing README is context for the new one - if [ -f "README.md" ]; then - README_SIZE=$(wc -c < README.md | tr -d ' ') - echo "Existing README.md: $README_SIZE bytes (used as context)" - else - echo "Existing README.md: none" - fi - - # 3. README, CONTRIBUTING and the docs index, one hub call each. - # Script already downloaded to /tmp/ in "Download pipeline scripts" - # run_stage records the outcome as stage4_status โ€” see its note in - # workflow-helpers.sh. The file count below is REPORTING, not a - # status: inferring "completed" from it meant a crashed run that left - # a previous commit's README standing reported success. - run_stage "Stage 4" "$STAGE4_TIMEOUT_HOURS" stage4_status node /tmp/generate-repo-docs.cjs - - # 4. Count results - REPO_DOCS=0 - for f in README.md CONTRIBUTING.md LICENSE.md SECURITY.md; do - if [ -f "$f" ]; then - SIZE=$(wc -c < "$f" | tr -d ' ') - echo "$f: $SIZE bytes" - REPO_DOCS=$((REPO_DOCS + 1)) - fi - done - - set_output "stage4_files" "$REPO_DOCS" - if [ "$REPO_DOCS" -gt 0 ]; then - echo "Repository docs present: $REPO_DOCS" - else - echo "::warning title=Stage 4 produced nothing::No repository docs (README.md, CONTRIBUTING.md, LICENSE.md, SECURITY.md) are present." - fi - - # ========================================================================= - # COMMIT STAGE 4 RESULTS (progressive pull request) - # ========================================================================= - # `!= ''` โ€” the run_stage commit rule, stated once at Stage 1. - - name: Commit and push stage 4 results - if: always() && steps.stage4.outputs.stage4_status != '' - id: commit-stage4 - env: - BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }} - STAGE4_FILES: ${{ steps.stage4.outputs.stage4_files }} - run: | - source /tmp/workflow-helpers.sh - - # .gitignore in each stage 4 managed directory - for managed_dir in docs/api docs/deployment docs/operations docs/cli; do - if [ -d "$managed_dir" ]; then - { - echo "# VoltAgent temp files" - echo "temp/" - echo "" - echo "# JSON intermediate files (except schema/config)" - echo "*.json" - echo "!*-schema.json" - echo "!*-config.json" - } > "$managed_dir/.gitignore" - echo "Wrote $managed_dir/.gitignore" - fi - done - - # Stage repository documentation files - # AGENTS.md is here as the backstop for a run whose Stage 2 commit did not happen. - for f in README.md CONTRIBUTING.md LICENSE.md SECURITY.md AGENTS.md; do - if [ -f "$f" ]; then - git add -f "$f" - fi - done - - # Stage Stage 4 managed directories - for managed_dir in docs/api docs/deployment docs/operations docs/cli; do - if [ -d "$managed_dir" ]; then - git add -f "$managed_dir/" 2>/dev/null || true - fi - done - - # Stage docs/README.md if exists - if [ -f "docs/README.md" ]; then - git add -f "docs/README.md" - fi - - # Check if there are changes - STAGED_COUNT=$(git diff --cached --name-only | wc -l) - - if [ "$STAGED_COUNT" -gt 0 ]; then - # Commit and push - git commit -m "docs: Stage 4 - Repository documentation ($STAGE4_FILES files) [skip ci]" - git push origin "$BRANCH_NAME" - - echo "Committed and pushed $STAGED_COUNT stage 4 files" - set_output "committed" "true" - else - echo "No stage 4 changes to commit" - set_output "committed" "false" - fi - - - name: Update pull request with stage 4 progress - if: always() && steps.commit-stage4.outputs.committed == 'true' && steps.create-initial-pr.outputs.pull-request-number - env: - GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} - PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }} - STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }} - STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }} - STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || 0 }} - STAGE4_FILES: ${{ steps.stage4.outputs.stage4_files || 0 }} - RUN_ID_VAR: ${{ env.RUN_ID }} - REPO_NAME: ${{ github.repository }} - run: | - source /tmp/workflow-helpers.sh - - # Update PR body - PR_BODY="## ๐Ÿฆฉ Flamingo Code Documentation: In Progress - - **Run ID:** \`$RUN_ID_VAR\` - **Status:** ๐Ÿ”„ Running... - - This PR is being updated as each documentation stage completes. - - ### Progress - - โœ… Stage 1: Inline Documentation - Completed ($STAGE1_FILES files) - - โœ… Stage 2: Architecture Analysis - Completed ($STAGE2_FILES files) - - โœ… Stage 3: Tutorial Generation - Completed ($STAGE3_FILES files) - - โœ… Stage 4: Repository Documentation - Completed ($STAGE4_FILES files) - - --- - ๐Ÿฆฉ Generated by [Flamingo Code Documentation](https://flamingo.run)" - - gh pr edit "$PR_NUMBER" --body "$PR_BODY" - - echo "Pull request #$PR_NUMBER updated with stage 4 progress" - - # ========================================================================= - # VALIDATE GENERATED MARKDOWN - # Warn-only validation (never blocks the pull request) - # ========================================================================= - - name: Validate generated Markdown - if: always() - continue-on-error: true # NEVER block PR - validation is warn-only - env: - DOCS_OUTPUT_DIR: ${{ env.DOCS_OUTPUT_PATH }} - # OSS Tenant Structure paths for validation - REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }} - DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }} - GETTING_STARTED_OUTPUT_PATH: ${{ env.GETTING_STARTED_OUTPUT_PATH }} - DEVELOPMENT_OUTPUT_PATH: ${{ env.DEVELOPMENT_OUTPUT_PATH }} - NODE_PATH: ${{ runner.temp }}/doc-orch-deps/node_modules - run: | - # Warn-only: findings are logged per directory and never fail the run. - echo "::group::Validate $DOCS_OUTPUT_DIR" - node /tmp/validate-markdown.js "$DOCS_OUTPUT_DIR" 2>&1 || true - echo "::endgroup::" - # Validate OSS Tenant Structure outputs - echo "::group::Validate $REFERENCE_OUTPUT_PATH" - node /tmp/validate-markdown.js "$REFERENCE_OUTPUT_PATH" 2>&1 || true - echo "::endgroup::" - echo "::group::Validate $GETTING_STARTED_OUTPUT_PATH" - node /tmp/validate-markdown.js "$GETTING_STARTED_OUTPUT_PATH" 2>&1 || true - echo "::endgroup::" - echo "::group::Validate $DEVELOPMENT_OUTPUT_PATH" - node /tmp/validate-markdown.js "$DEVELOPMENT_OUTPUT_PATH" 2>&1 || true - echo "::endgroup::" - - - name: Report stage 4 progress - if: always() && env.HUB_BASE_URL != '' - continue-on-error: true - env: - WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }} - WORKFLOW_RUN_ID: ${{ github.run_id }} - WORKFLOW_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} - STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }} - STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }} - STAGE2_STATUS: ${{ steps.stage2.outputs.stage2_status || steps.stage2_alt.outputs.stage2_status || 'skipped' }} - STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }} - STAGE3_STATUS: ${{ steps.stage3.outputs.stage3_status || 'skipped' }} - STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || 0 }} - STAGE4_STATUS: ${{ steps.stage4.outputs.stage4_status || 'skipped' }} - STAGE4_FILES: ${{ steps.stage4.outputs.stage4_files || 0 }} - PR_URL: ${{ steps.create-initial-pr.outputs.pull-request-url }} - PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }} - run: | - source /tmp/workflow-helpers.sh - - CALLBACK_URL="${HUB_BASE_URL}/api/code-documentation/webhook" - echo "Reporting stage 4 (repository docs): $STAGE4_STATUS, $STAGE4_FILES files" - report_stage_progress "$CALLBACK_URL" "$WEBHOOK_SECRET" "$RUN_ID" "$REPO_ID" \ - "$WORKFLOW_RUN_ID" "$WORKFLOW_URL" "creating-pr" \ - "$STAGE1_STATUS" "$STAGE1_FILES" "$STAGE2_STATUS" "$STAGE2_FILES" "$STAGE3_STATUS" "$STAGE3_FILES" \ - "$STAGE4_STATUS" "$STAGE4_FILES" "$PR_URL" "$PR_NUMBER" - - # ========================================================================= - # FINALIZE THE PULL REQUEST - # ========================================================================= - - name: Clean up temporary files - run: | - source /tmp/workflow-helpers.sh - - # Remove stats files - cleanup_path ".doc-stage1-stats.json" - cleanup_path ".doc-stage3-stats.json" - - # Remove run status file (created for the initial PR). - # Legacy name kept so branches started before the rename still clean up. - cleanup_path ".flamingo-ai-technical-writer-status.md" - cleanup_path ".doc-pipeline-status.md" - - # Note: .doc-orchestrator-source-files.txt is now in /tmp/ (auto-cleanup) - - # Remove npm artifacts (installed for scripts) - cleanup_path "node_modules" - cleanup_path "package.json" - cleanup_path "package-lock.json" - - # NOTE: /tmp/workflow-helpers.sh is removed by "Report run result" - - - name: Stage remaining docs and detect changes - id: stage-docs - run: | - source /tmp/workflow-helpers.sh - echo "Docs root: $DOCS_OUTPUT_PATH" - echo "Stage 2: $REFERENCE_OUTPUT_PATH (reference), $DIAGRAMS_OUTPUT_PATH (diagrams)" - echo "Stage 3: $GETTING_STARTED_OUTPUT_PATH (getting started), $DEVELOPMENT_OUTPUT_PATH (development)" - - # Count untracked/modified files before staging - BEFORE_COUNT=$(git status --porcelain | wc -l) - echo "Changed files in the working tree: $BEFORE_COUNT" - - # Stage ALL .md and .mmd files anywhere in the repo (for inline docs generated next to source files) - # This catches Stage 1 inline docs (hidden: .FileName.md), Stage 2 reference/diagrams, Stage 3 tutorials, and Stage 4 repo docs - # Find all .md and .mmd files recursively, including hidden files (.*.md) - # Includes README.md, CONTRIBUTING.md, LICENSE.md, SECURITY.md from Stage 4 - # Includes .mmd Mermaid diagram files from Stage 2 (CodeWiki/Claude architecture) - find . \( -name "*.md" -o -name ".*.md" -o -name "*.mmd" \) -type f \ - -not -path "./node_modules/*" \ - -not -path "./.git/*" \ - -not -name "CHANGELOG.md" \ - -exec git add -f {} \; 2>/dev/null || true - - echo "::group::Markdown and Mermaid files in the checkout (first 100)" - find . \( -name "*.md" -o -name ".*.md" -o -name "*.mmd" \) -type f \ - -not -path "./node_modules/*" \ - -not -path "./.git/*" \ - -not -name "CHANGELOG.md" | head -100 - echo "::endgroup::" - - # Count staged files - STAGED_COUNT=$(git diff --cached --name-only | wc -l) - echo "Staged files: $STAGED_COUNT" - set_output "staged_count" "$STAGED_COUNT" - - echo "::group::Staged files (first 50)" - git diff --cached --name-only | head -50 - echo "::endgroup::" - - # "Changes" means the BRANCH differs from the documented source head, not - # that this final sweep found something left to stage: every stage commits - # its own output as it goes, so a run whose stages all committed (inline - # docs, README) left nothing here and was reported as no_changes while its - # pull request held twenty files (openframe-saas-mobile run 35302244804). - # The status file is the run's own bookkeeping, never a documentation change. - if [ "$STAGED_COUNT" -eq "0" ] && git diff --quiet "$SOURCE_HEAD_SHA" HEAD -- . ':!.flamingo-ai-technical-writer-status.md'; then - echo "::notice title=No documentation changes::Nothing left to stage, and the branch holds no documentation change against the source head." - set_output "has_changes" "false" - else - set_output "has_changes" "true" - fi - - - name: Mark pull request complete - if: steps.create-initial-pr.outputs.pull-request-number - env: - GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} - PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }} - STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }} - STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }} - STAGE2_STATUS: ${{ steps.stage2.outputs.stage2_status || steps.stage2_alt.outputs.stage2_status || 'skipped' }} - STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }} - STAGE3_STATUS: ${{ steps.stage3.outputs.stage3_status || 'skipped' }} - STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || 0 }} - STAGE4_STATUS: ${{ steps.stage4.outputs.stage4_status || 'skipped' }} - STAGE4_FILES: ${{ steps.stage4.outputs.stage4_files || 0 }} - RUN_ID_VAR: ${{ env.RUN_ID }} - REPO_NAME: ${{ github.repository }} - run: | - source /tmp/workflow-helpers.sh - - # Remove "in-progress" label - gh pr edit "$PR_NUMBER" --remove-label "in-progress" || true - - # Update title to remove [IN PROGRESS] - gh pr edit "$PR_NUMBER" --title "๐Ÿฆฉ Flamingo Code Documentation" - - # Update body with final results - PR_BODY="## ๐Ÿฆฉ Flamingo Code Documentation: Complete - - **Run ID:** \`$RUN_ID_VAR\` - - ### Stage 1: Inline Documentation - - Status: $STAGE1_STATUS - - Files generated: $STAGE1_FILES - - Generated .md files next to source classes explaining their purpose - - ### Stage 2: Architecture Analysis - - Status: $STAGE2_STATUS - - Files generated: $STAGE2_FILES - - Architecture overview and module documentation - - ### Stage 3: AI Tutorial Generator - - Status: $STAGE3_STATUS - - Files generated: $STAGE3_FILES - - Getting started guides and how-to tutorials - - ### Stage 4: Repository Documentation - - Status: $STAGE4_STATUS - - Files generated: $STAGE4_FILES - - README.md, CONTRIBUTING.md, LICENSE.md, SECURITY.md - - --- - - **Review checklist:** - - [ ] Check generated inline docs for accuracy - - [ ] Review architecture documentation - - [ ] Test code examples in tutorials - - [ ] Review README.md and CONTRIBUTING.md updates - - --- - ๐Ÿฆฉ Generated by [Flamingo Code Documentation](https://flamingo.run)" - - gh pr edit "$PR_NUMBER" --body "$PR_BODY" - - echo "Pull request #$PR_NUMBER marked complete" - - # ========================================================================= - # REPORT RUN RESULT: the terminal callback. Never reports "running". - # ========================================================================= - - name: Report run result - if: always() && env.HUB_BASE_URL != '' - continue-on-error: true # Don't fail the workflow if callback fails - env: - # SECURITY: Pass secret per-step with inline masking - WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }} - WORKFLOW_RUN_ID: ${{ github.run_id }} - WORKFLOW_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} - HAS_CHANGES: ${{ steps.stage-docs.outputs.has_changes }} - JOB_STATUS: ${{ job.status }} - PR_URL: ${{ steps.create-initial-pr.outputs.pull-request-url }} - PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number || 'null' }} - SAFE_RUN_ID: ${{ steps.branch-name-early.outputs.safe_run_id }} - # Single source of truth for the branch name (was rebuilt by hand below, - # which silently drifted from "Create docs branch" on every rename) - BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }} - STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }} - STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }} - # Stage 2: Check both CodeWiki and Claude alternative, mark as failed if step failed - STAGE2_STATUS: ${{ steps.stage2.outcome == 'failure' && 'failed' || steps.stage2_alt.outcome == 'failure' && 'failed' || steps.stage2.outputs.stage2_status || steps.stage2_alt.outputs.stage2_status || 'skipped' }} - STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }} - STAGE3_STATUS: ${{ steps.stage3.outputs.stage3_status || 'skipped' }} - STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || 0 }} - STAGE4_STATUS: ${{ steps.stage4.outputs.stage4_status || 'skipped' }} - STAGE4_FILES: ${{ steps.stage4.outputs.stage4_files || 0 }} - # Track if critical steps failed (continue-on-error: false steps) - STAGE2_OUTCOME: ${{ steps.stage2.outcome || 'skipped' }} - CODEWIKI_INSTALL_OUTCOME: ${{ steps.codewiki_install.outcome || 'skipped' }} - CODEWIKI_CONFIG_OUTCOME: ${{ steps.codewiki_config.outcome || 'skipped' }} - run: | - # Bootstrap-failure fallback (shared failure-net standard with the - # code-review workflow): if workflow-helpers.sh never downloaded, no - # helper exists to report the failure โ€” a minimal guarded curl posts - # it so the hub's run row fails NOW instead of waiting for the reaper. - if [ ! -f /tmp/workflow-helpers.sh ]; then - echo "::error title=Script bootstrap failed::workflow-helpers.sh was never downloaded from the hub; reporting the failure with a minimal callback." - # The bearer goes through a 0600 config file, never argv โ€” see - # curlAuthPreamble in lib/config/workflow-scripts-bootstrap.ts. - CURL_CFG=$(mktemp) && chmod 600 "$CURL_CFG" - trap 'rm -f "$CURL_CFG"' EXIT - printf 'header = "Authorization: Bearer %s"\n' "$WEBHOOK_SECRET" > "$CURL_CFG" - curl -sS --max-time 30 -K "$CURL_CFG" -X POST "${HUB_BASE_URL}/api/code-documentation/webhook" \ - -H "Content-Type: application/json" \ - -d "{\"run_id\":\"$RUN_ID\",\"repo_id\":\"$REPO_ID\",\"status\":\"failure\",\"workflow_run_id\":$WORKFLOW_RUN_ID,\"workflow_url\":\"$WORKFLOW_URL\",\"error\":\"Script bootstrap failed: workflow-helpers.sh never downloaded from the hub.\"}" || true - exit 1 - fi - source /tmp/workflow-helpers.sh - - CALLBACK_URL="${HUB_BASE_URL}/api/code-documentation/webhook" - - echo "::group::Inputs to the final status" - echo "JOB_STATUS=$JOB_STATUS" - echo "HAS_CHANGES=$HAS_CHANGES" - echo "STAGE2_OUTCOME=$STAGE2_OUTCOME" - echo "CODEWIKI_INSTALL_OUTCOME=$CODEWIKI_INSTALL_OUTCOME" - echo "CODEWIKI_CONFIG_OUTCOME=$CODEWIKI_CONFIG_OUTCOME" - echo "::endgroup::" - - # CRITICAL: Determine final status - NEVER return "running" - # Default to failure, only set success if everything checks out - STATUS="failure" - - # Check for cancelled job first - if [ "$JOB_STATUS" = "cancelled" ]; then - STATUS="cancelled" - REASON="the workflow run was cancelled" - # Check if critical stage 2 (CodeWiki) failed - this has continue-on-error: false - elif [ "$STAGE2_OUTCOME" = "failure" ]; then - STATUS="failure" - REASON="the CodeWiki stage failed" - # Check if CodeWiki installation failed - elif [ "$CODEWIKI_INSTALL_OUTCOME" = "failure" ]; then - STATUS="failure" - REASON="the CodeWiki installation failed" - # Check if CodeWiki configuration failed - elif [ "$CODEWIKI_CONFIG_OUTCOME" = "failure" ]; then - STATUS="failure" - REASON="the CodeWiki configuration failed" - # Check overall job status - elif [ "$JOB_STATUS" != "success" ]; then - STATUS="failure" - REASON="the job status is $JOB_STATUS" - # Check if we have any documentation changes - elif [ "$HAS_CHANGES" != "true" ]; then - STATUS="no_changes" - REASON="no documentation changed" - else - STATUS="success" - REASON="documentation generated" - fi - - # SAFETY CHECK: Ensure status is NEVER "running" - if [ "$STATUS" = "running" ] || [ -z "$STATUS" ]; then - echo "::warning::Computed status '$STATUS' is not terminal; reporting failure instead." - STATUS="failure" - REASON="no terminal status could be determined" - fi - - case "$STATUS" in - success) echo "Final status: success ($REASON)" ;; - no_changes) echo "::notice title=Code Documentation: no changes::Final status: no_changes ($REASON)." ;; - cancelled) echo "::warning title=Code Documentation: cancelled::Final status: cancelled ($REASON)." ;; - *) echo "::error title=Code Documentation: $STATUS::Final status: $STATUS ($REASON)." ;; - esac - - # Branch actually created by "Create docs branch". Falls back to the same - # formula only when that step never ran (this step is `if: always()`). - SAFE_BRANCH="${BRANCH_NAME:-docs/flamingo-ai-technical-writer-$SAFE_RUN_ID}" - - # Send final webhook using helper function - report_final_status "$CALLBACK_URL" "$WEBHOOK_SECRET" "$RUN_ID" "$REPO_ID" \ - "$WORKFLOW_RUN_ID" "$WORKFLOW_URL" "$STATUS" "$PR_URL" "$PR_NUMBER" "$SAFE_BRANCH" \ - "$STAGE1_STATUS" "$STAGE1_FILES" "$STAGE2_STATUS" "$STAGE2_FILES" "$STAGE3_STATUS" "$STAGE3_FILES" \ - "$STAGE4_STATUS" "$STAGE4_FILES" - - # Final cleanup: remove workflow helpers file - cleanup_path "/tmp/workflow-helpers.sh" - - # The run's job summary (the Actions run page). Values reach the script - # through env, never inline expressions. - - name: Write job summary - if: always() - env: - STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }} - STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || '0' }} - STAGE2_STATUS: ${{ steps.stage2.outputs.stage2_status || steps.stage2_alt.outputs.stage2_status || 'skipped' }} - STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || '0' }} - STAGE2_ENGINE: ${{ steps.detect_language.outputs.codewiki_supported == 'true' && 'CodeWiki' || steps.detect_language.outputs.codewiki_supported == 'false' && 'Claude' || 'engine not chosen' }} - STAGE3_STATUS: ${{ steps.stage3.outputs.stage3_status || 'skipped' }} - STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || '0' }} - STAGE4_STATUS: ${{ steps.stage4.outputs.stage4_status || 'skipped' }} - STAGE4_FILES: ${{ steps.stage4.outputs.stage4_files || '0' }} - GRAPH_OUTCOME: ${{ steps.graph.outcome || 'skipped' }} - PRIMARY_LANGUAGE: ${{ steps.detect_language.outputs.primary_language }} - PR_URL: ${{ steps.create-initial-pr.outputs.pull-request-url }} - run: | - { - echo "## ๐Ÿฆฉ Flamingo Code Documentation" - echo "" - echo "Run \`$RUN_ID\` ยท source branch \`$SOURCE_BRANCH\` ยท primary language ${PRIMARY_LANGUAGE:-unknown}" - echo "" - echo "| Stage | Status | Files |" - echo "|-------|--------|-------|" - echo "| 0 ยท Code graph | $GRAPH_OUTCOME | โ€“ |" - echo "| 1 ยท Inline docs | $STAGE1_STATUS | $STAGE1_FILES |" - echo "| 2 ยท Reference docs ($STAGE2_ENGINE) | $STAGE2_STATUS | $STAGE2_FILES |" - echo "| 3 ยท Tutorials | $STAGE3_STATUS | $STAGE3_FILES |" - echo "| 4 ยท Repository docs | $STAGE4_STATUS | $STAGE4_FILES |" - echo "" - if [ -n "$PR_URL" ]; then - echo "**Pull request:** $PR_URL" - else - echo "No pull request was opened." - fi - } >> "$GITHUB_STEP_SUMMARY"