Skip to content

fix(CODEWIKI-005-2): 2 review findings across 2 files #63

fix(CODEWIKI-005-2): 2 review findings across 2 files

fix(CODEWIKI-005-2): 2 review findings across 2 files #63

# 🦩 Flamingo Code Documentation (doc-orchestrator pipeline)
# ===========================
# This GitHub Actions workflow is triggered by the multi-platform-hub to generate
# comprehensive documentation for this repository and create a PR with the results.
#
# 4-Stage Pipeline:
# 1. Inline Documentation - Generate .md files next to source classes
# 2. CodeWiki Analysis - Architecture overview and Mermaid diagrams
# 3. AI Tutorial Generator - VoltAgent-powered tutorial generation with tool-based exploration
# 4. Repository Documentation - Generate/update README.md, CONTRIBUTING.md; copy LICENSE.md, SECURITY.md
#
# NOTE: Stage 3 uses the VoltAgent framework (https://voltagent.dev/) for agentic
# document generation with three tools: list_docs, read_doc_file, write_tutorial_doc.
# NOTE: Stage 4 uses VoltAgent to update README with OpenFrame structure while preserving existing content.
#
# Installation:
# 1. Copy this file to .github/workflows/doc-orchestrator.yml in your target repository
# 2. Configure these GitHub Secrets in the target repository (Settings > Secrets):
# - ANTHROPIC_API_KEY: For the VoltAgent stages (3 & 4); Stages 1 and 2 call Claude through the hub
# - OPENAI_API_KEY: For CodeWiki (Stage 2)
# - DOC_ORCH_WEBHOOK_SECRET: For callback authentication
# - DOC_ORCH_GITHUB_PAT: (Optional) For private dependency access
# - YOUTUBE_API_KEY: (Optional) For YouTube video embedding in Stage 3 & 4
name: 🦩 Flamingo Code Documentation
on:
# Push trigger — two things ride it. It registers the workflow with GitHub
# Actions (required for the workflow_dispatch API), and on the repository's
# DEFAULT branch it runs the `code-graph` job below, which re-indexes the
# code graph the hub serves to the code reviewer and to the documentation
# stages. The documentation pipeline itself NEVER runs on push (see its
# `if:`). Documentation and markdown are ignored on purpose: a docs PR
# merging must not rebuild a graph that only source files can change.
push:
paths-ignore:
- 'docs/**'
- '**.md'
repository_dispatch:
# `doc-orchestrator` runs the documentation pipeline. `flamingo-code-graph`
# (CODE_GRAPH_DISPATCH_EVENT_TYPE in lib/config/code-graph-workflow.ts)
# runs ONLY the graph job — the hub's reconcile job sends it when a repo's
# graph is missing or stale.
types: [doc-orchestrator, flamingo-code-graph]
workflow_dispatch:
# ═══════════════════════════════════════════════════════════════════════════
# GENERATED FROM SINGLE SOURCE OF TRUTH: lib/config/doc-orchestrator-params.ts
# This section is auto-generated at runtime when creating workflow PRs
# ═══════════════════════════════════════════════════════════════════════════
inputs:
# Individual parameters
run_id:
description: 'Unique execution ID'
required: true
repo_id:
description: 'Repository ID from database'
required: true
hub_base_url:
description: 'Hub base URL (e.g., https://product-hub.flamingo.so)'
required: true
stages:
description: 'Pipeline stages to execute'
required: true
dependencies:
description: 'Comma-separated dependency repos'
required: false
default: ''
source_branch:
description: 'Branch to analyze code from'
required: true
source_files_limit:
description: 'Max source files to process (0 = unlimited)'
required: true
claude_model:
description: 'Claude model ID for Stage 1/3/4 + Stage 2 Claude-arch fallback (SSOT from hub)'
required: true
codewiki_config:
description: 'Complete CodeWiki configuration (per-phase models, engine, stages, depth)'
required: true
output_paths:
description: 'Output paths configuration'
required: true
timeout:
description: 'Timeout in hours'
required: true
youtube_config:
description: 'YouTube integration configuration (channels only - API key in secrets)'
required: true
readme_config:
description: 'README logo configuration'
required: true
custom_repo_instructions:
description: 'Custom AI instructions'
required: false
default: ''
external_repos:
description: 'External repos JSON'
required: false
default: '[]'
stage_count:
description: 'Total number of pipeline stages'
required: false
default: '4'
graph_only:
description: 'true = run only the code-graph job (no documentation run)'
required: false
default: 'false'
# ═══════════════════════════════════════════════════════════════════════════
# END GENERATED SECTION
# ═══════════════════════════════════════════════════════════════════════════
env:
# SECURITY: Only NON-SENSITIVE variables in job-level env
# Secrets are passed per-step to avoid exposure in job setup logs
# Run configuration (non-sensitive)
RUN_ID: ${{ github.event.client_payload.run_id || github.event.inputs.run_id || github.run_id }}
REPO_ID: ${{ github.event.client_payload.repo_id || github.event.inputs.repo_id || '' }}
# A push event carries no payload, so the graph job falls back to the org
# Actions variable — the same fallback the code-review workflow uses.
HUB_BASE_URL: ${{ github.event.client_payload.hub_base_url || github.event.inputs.hub_base_url || vars.FLAMINGO_HUB_BASE_URL || '' }}
# 'true' = run only the code-graph job (a manual workflow_dispatch; the hub's
# own documentation dispatches always send 'false').
GRAPH_ONLY: ${{ github.event.client_payload.graph_only || github.event.inputs.graph_only || 'false' }}
# No literal fallback: the four-stage list lives in DOC_ORCHESTRATOR_STAGES on the
# hub and is lifted into the payload per repo. A literal here would silently
# restore all four stages on a payload gap — and Clean Slate would still have
# wiped the docs tree first, so the run would delete docs and regenerate nothing.
STAGES: ${{ github.event.client_payload.stages || github.event.inputs.stages || '' }}
DEPENDENCIES: ${{ github.event.client_payload.dependencies || github.event.inputs.dependencies || '' }}
STAGE_COUNT: ${{ github.event.client_payload.stage_count || github.event.inputs.stage_count || '4' }}
# Branch to checkout for code analysis (github_branch from repo config)
SOURCE_BRANCH: ${{ github.event.client_payload.source_branch || github.event.inputs.source_branch || 'main' }}
# Debug/testing: limit total source files to analyze (0=unlimited)
# Files beyond this limit are DELETED - all stages then process remaining files
SOURCE_FILES_LIMIT: ${{ github.event.client_payload.source_files_limit || github.event.inputs.source_files_limit || '0' }}
# Claude model SSOT: passed in from hub's CLAUDE_MODELS.SONNET constant
# (see `lib/constants/ai-models.ts DOC_ORCHESTRATOR_DEFAULT_MODEL`).
# Stage 1/3/4 generator scripts + Stage 2 Claude-arch fallback all read
# this env var. NO literal model string is allowed in any shipped
# template script — scripts throw if CLAUDE_MODEL is empty. Re-run
# "Setup Workflow" after bumping the hub-side constant to propagate.
# There is deliberately NO companion request-shape env: Stage 1/2 call Claude
# through the hub (/api/ci/claude), which resolves the model's request shape
# itself, and Stage 3/4 (VoltAgent) run adaptive thinking by design.
CLAUDE_MODEL: ${{ github.event.client_payload.claude_model || github.event.inputs.claude_model || '' }}
# =============================================================================
# JSON-grouped parameters to stay under GitHub Actions 25-parameter limit
# These are parsed early in the workflow to extract individual values
# =============================================================================
# NOTE: these fall back to EMPTY, not '{}'. An empty-object default made the
# `[ -z ... ]` presence checks below unreachable, so a missing payload silently
# produced `null` for every jq lookup and propagated as `--cluster-model null`.
# The hub always sends a complete, deep-merged blob (buildPayloadFromRepo).
CODEWIKI_CONFIG_JSON: ${{ github.event.client_payload.codewiki_config || github.event.inputs.codewiki_config || '' }}
OUTPUT_PATHS_JSON: ${{ github.event.client_payload.output_paths || github.event.inputs.output_paths || '' }}
# Stage timeouts (in hours) — single source of truth, downstream steps reference
# `env.STAGE_TIMEOUT_HOURS` directly. Stage 4 is the exception (1h vs 24h cap).
STAGE_TIMEOUT_HOURS: ${{ github.event.client_payload.timeout || github.event.inputs.timeout || '24' }}
# Aliases preserved for downstream step env: keys (they reference these names
# by string). All resolve to the same single source.
STAGE1_TIMEOUT_HOURS: ${{ github.event.client_payload.timeout || github.event.inputs.timeout || '24' }}
# Stage 1 incremental push: commit + push the PR branch every N generated inline
# docs. A 24h stage that gets cancelled used to lose ALL of its work because the
# only commit happened after the generator returned. Bound the loss to N files.
STAGE1_PUSH_INTERVAL: ${{ github.event.client_payload.stage1_push_interval || '100' }}
STAGE2_TIMEOUT_HOURS: ${{ github.event.client_payload.timeout || github.event.inputs.timeout || '24' }}
STAGE3_TIMEOUT_HOURS: ${{ github.event.client_payload.timeout || github.event.inputs.timeout || '24' }}
# Stage 4: Repository Documentation
TEMPLATE_REPO: ${{ github.event.client_payload.template_repo || 'flamingo-stack/openframe-oss-tenant' }}
TEMPLATE_BRANCH: ${{ github.event.client_payload.template_branch || 'main' }}
STAGE4_TIMEOUT_HOURS: ${{ github.event.client_payload.stage4_timeout || '1' }}
# YouTube Integration (Stage 3 + Stage 4) - JSONB configuration (API key from secrets)
YOUTUBE_CONFIG_JSON: ${{ github.event.client_payload.youtube_config || github.event.inputs.youtube_config || '' }}
# README Configuration - JSONB configuration for logo branding
README_CONFIG_JSON: ${{ github.event.client_payload.readme_config || github.event.inputs.readme_config || '' }}
# Custom AI Instructions (All Stages) - Repository-specific instructions for AI generation
CUSTOM_REPO_INSTRUCTIONS: ${{ github.event.client_payload.custom_repo_instructions || github.event.inputs.custom_repo_instructions || '' }}
# External Repositories - JSON array of external repo configurations
EXTERNAL_REPOS: ${{ github.event.client_payload.external_repos || github.event.inputs.external_repos || '[]' }}
# ========================================================================
# Repository Context - CRITICAL for preventing AI URL hallucinations
# These values are passed to ALL AI prompts to ensure correct GitHub URLs
# ========================================================================
GITHUB_REPOSITORY: ${{ github.repository }} # e.g., "flamingo-stack/openframe-oss-tenant"
GITHUB_REPOSITORY_OWNER: ${{ github.repository_owner }} # e.g., "flamingo-stack"
GITHUB_SERVER_URL: ${{ github.server_url }} # e.g., "https://github.com"
# Analysis Exclusions - Complete array of glob patterns to exclude from repository analysis
# Prevents analyzing orchestrator frontend, build artifacts, dependencies, generated docs
EXCLUDED_PATHS: '**/node_modules/**,**/.git/**,**/target/**,**/dist/**,**/build/**,**/.next/**,**/out/**,**/coverage/**,**/vendor/**,**/.yalc/**,**/.turbo/**,**/.gradle/**,**/__pycache__/**,**/.terraform/**,**/.venv/**,**/venv/**,**/multi-platform-hub/**,**/deps-*/**'
README_LOGO_ALT: 'OpenFrame Logo'
jobs:
# ===========================================================================
# CODE GRAPH — deterministic, no model call. Tags every public symbol,
# import and manifest of the checkout (code-graph-build.mjs) and uploads the
# result to the hub, which promotes a default-branch snapshot to `live` and
# serves it to the code reviewer (consumers of a symbol a PR removes) and to
# the documentation stages (the derived ecosystem.md). Runs on every push to
# the default branch, on the hub's `flamingo-code-graph` re-dispatch, and on
# a manual workflow_dispatch with graph_only=true. There is no webhook
# callback: the upload IS the report. The same build also runs as stage 0 of
# a full documentation run (inside doc-pipeline, below).
# ===========================================================================
code-graph:
runs-on: ubuntu-latest
timeout-minutes: 20
permissions:
contents: read
concurrency:
group: flamingo-code-graph-${{ github.repository }}
cancel-in-progress: true
if: >-
(github.event_name == 'push' && github.ref == format('refs/heads/{0}', github.event.repository.default_branch))
|| github.event.action == 'flamingo-code-graph'
|| (github.event_name == 'workflow_dispatch' && github.event.inputs.graph_only == 'true')
steps:
# Fail LOUD, not silent: a push on a repo whose org never set
# FLAMINGO_HUB_BASE_URL would otherwise curl an empty origin and die with
# an unrelated error. Also normalizes a trailing slash ONCE.
- name: Validate configuration
run: |
if [ -z "$HUB_BASE_URL" ]; then
echo "::error::HUB_BASE_URL is empty — set the org Actions variable FLAMINGO_HUB_BASE_URL (or pass hub_base_url in the dispatch payload)."
exit 1
fi
echo "HUB_BASE_URL=${HUB_BASE_URL%/}" >> "$GITHUB_ENV"
# The shared script bootstrap (byte-mirrored from workflow-scripts-bootstrap.ts,
# asserted by the build gate). BOTH graph scripts are downloaded: the builder
# imports ./code-graph-lib.mjs from its own directory.
- name: Download graph scripts
env:
WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }}
run: |
# Function to download and verify script
SCRIPT_MANIFEST=/tmp/flamingo-script-manifest.json
# WEBHOOK_SECRET reaches curl through a 0600 config file, never argv — see
# curlAuthPreamble, which always traps the removal.
CURL_CFG=$(mktemp) && chmod 600 "$CURL_CFG"
trap 'rm -f "$CURL_CFG"' EXIT
printf 'header = "Authorization: Bearer %s"\n' "$WEBHOOK_SECRET" > "$CURL_CFG"
# The canonical scripts surface and the pre-rename one. load_script_manifest
# picks whichever this deployment actually serves and pins SCRIPTS_BASE_URL.
CI_SCRIPTS_URL="${HUB_BASE_URL%/}/api/ci/scripts"
LEGACY_SCRIPTS_URL="${HUB_BASE_URL%/}/api/doc-orchestrator/scripts"
# _try_manifest <base> — 0 loaded, 1 no manifest surface there, 2 fatal.
# The manifest is asked for ONE group: its keys are the files to download.
_try_manifest() {
local base="$1" code
code=$(curl -sS -w '%{http_code}' -o "$SCRIPT_MANIFEST" \
-K "$CURL_CFG" \
"$base/manifest.json?group=$SCRIPT_GROUP") || code="000"
if [ "$code" = "404" ]; then rm -f "$SCRIPT_MANIFEST"; return 1; fi
if [ "$code" != "200" ]; then
echo "❌ manifest request to $base failed (HTTP $code)"
rm -f "$SCRIPT_MANIFEST"
return 2
fi
# The digests are the TOP-LEVEL object. successResponse is the standard
# emitter but it does NOT add a wrapper — it is NextResponse.json(data)
# plus the no-store header — so there is no .data to reach through.
# A 200 that is not a manifest is how a hub which does not serve this path
# answers (the proxy rewrites unknown routes and returns HTML), so it
# means "wrong surface", not "corrupt".
if ! jq -e 'type == "object" and length > 0 and (to_entries | all(.value | type == "string"))' "$SCRIPT_MANIFEST" >/dev/null 2>&1; then
rm -f "$SCRIPT_MANIFEST"
return 1
fi
return 0
}
# load_script_manifest <group>
load_script_manifest() {
SCRIPT_GROUP="$1"
# The scripts surface was renamed from /api/doc-orchestrator/scripts to the
# pipeline-neutral /api/ci/scripts (it always served BOTH pipelines). The
# workflow file ships in the repo and the routes ship with the deployment,
# so the two are one version apart in BOTH directions across the rollout.
# Probe the canonical surface, fall back to the legacy one, and let the
# winner decide SCRIPTS_BASE_URL for every download that follows.
# "cmd; rc=$?" dies under the set -euo pipefail these steps run with —
# errexit fires before rc is read and the step ends with NO output. And
# "if ! cmd; then rc=$?" is worse: inside the branch $? is the status of
# the NEGATION (0), so every failure reads as success. "|| rc=$?" is the
# one form that both suppresses errexit and preserves the real code.
local rc=0
_try_manifest "$CI_SCRIPTS_URL" || rc=$?
if [ "$rc" = "0" ]; then
SCRIPTS_BASE_URL="$CI_SCRIPTS_URL"
echo "✅ script manifest loaded ($(jq -r 'length' "$SCRIPT_MANIFEST") scripts)"
return 0
fi
if [ "$rc" = "2" ]; then exit 1; fi
echo "::warning::this hub does not serve $CI_SCRIPTS_URL — falling back to the legacy $LEGACY_SCRIPTS_URL; it predates the rename"
SCRIPTS_BASE_URL="$LEGACY_SCRIPTS_URL"
rc=0
_try_manifest "$LEGACY_SCRIPTS_URL" || rc=$?
if [ "$rc" = "0" ]; then
echo "✅ script manifest loaded ($(jq -r 'length' "$SCRIPT_MANIFEST") scripts)"
return 0
fi
if [ "$rc" = "2" ]; then exit 1; fi
# Neither surface published a manifest: a hub older than the manifest
# itself. The manifest is the file list, so there is nothing to download.
echo "❌ no script manifest on this hub ($HUB_BASE_URL): it cannot name the $SCRIPT_GROUP scripts. Redeploy the hub."
exit 1
}
# download_script_group <group> — the hub names the files, this workflow
# names only the group. Downloads every script of the group, in served order.
download_script_group() {
load_script_manifest "$1"
local name
# The loop runs in THIS shell (no pipe), so a failed download exits the step.
while IFS= read -r name; do
download_and_verify "$name"
done < <(jq -r 'keys_unsorted[]' "$SCRIPT_MANIFEST")
}
download_and_verify() {
local script_name="$1"
local output_path="/tmp/$script_name"
local expected_hash
expected_hash=$(jq -r --arg n "$script_name" '.[$n] // empty' "$SCRIPT_MANIFEST")
if [ -z "$expected_hash" ]; then
echo "❌ $script_name is not in the server's script manifest!"
echo " The hub serves no such script, or it failed to read on the server."
exit 1
fi
if ! printf '%s' "$expected_hash" | grep -Eq '^[0-9a-f]{64}$'; then
echo "❌ the manifest entry for $script_name is not a SHA-256 digest — refusing to run it."
exit 1
fi
curl -fsSL "$SCRIPTS_BASE_URL/$script_name" \
-K "$CURL_CFG" \
-o "$output_path"
local actual_hash=$(shasum -a 256 "$output_path" | cut -d' ' -f1)
if [ "$actual_hash" != "$expected_hash" ]; then
echo "❌ HASH MISMATCH for $script_name!"
echo " Expected: $expected_hash"
echo " Actual: $actual_hash"
echo " The download was corrupted in transit — both values come from the same deployment."
exit 1
fi
# Make shell scripts executable
if [[ "$script_name" == *.sh ]]; then
chmod +x "$output_path"
fi
echo "✅ $script_name verified (hash: ${actual_hash:0:16}...)"
}
# Digests AND the file list come from the deployment serving the bytes,
# not from this file: the step names a group (SCRIPT_GROUPS in the hub's
# lib/config/ci-script-catalog.ts) and downloads what the hub lists for it.
download_script_group "code-graph"
- name: Checkout Repository (graph)
uses: actions/checkout@v5
with:
fetch-depth: 1
persist-credentials: false
- name: Setup Node.js (graph)
uses: actions/setup-node@v4
with:
node-version: '22'
# The command is CODE_GRAPH_INSTALL_COMMAND (lib/config/code-graph-workflow.ts),
# the one spelling both workflows use: pinned wasm tree-sitter + grammars +
# yaml into an isolated tree under RUNNER_TEMP, exported as CODE_GRAPH_DEPS_DIR.
- name: Install graph dependencies
run: mkdir -p "$RUNNER_TEMP/code-graph-deps" && cd "$RUNNER_TEMP/code-graph-deps" && printf '%s' '{"name":"code-graph-deps","version":"1.0.0","private":true,"dependencies":{"web-tree-sitter":"0.27.0","@vscode/tree-sitter-wasm":"0.3.1","yaml":"2.9.1"}}' > package.json && printf '%s' '{"name":"code-graph-deps","version":"1.0.0","lockfileVersion":3,"requires":true,"packages":{"":{"name":"code-graph-deps","version":"1.0.0","dependencies":{"web-tree-sitter":"0.27.0","@vscode/tree-sitter-wasm":"0.3.1","yaml":"2.9.1"}},"node_modules/web-tree-sitter":{"version":"0.27.0","resolved":"https://registry.npmjs.org/web-tree-sitter/-/web-tree-sitter-0.27.0.tgz","integrity":"sha512-XK08gj6RwTMQatAG7uVRP8MunqotL/XC19vHgkSPKmELgbGPBj4ECvB8haHOUnyj6ls2B8t42UTro14zxGgAHg=="},"node_modules/@vscode/tree-sitter-wasm":{"version":"0.3.1","resolved":"https://registry.npmjs.org/@vscode/tree-sitter-wasm/-/tree-sitter-wasm-0.3.1.tgz","integrity":"sha512-RJFoomET6FajjG511fmQxeBQfU6M24a0aFZPqpid+ttIxanWf1VGytBG0UmsGjt07qmIPJS8U31D+aecuCucsQ=="},"node_modules/yaml":{"version":"2.9.1","resolved":"https://registry.npmjs.org/yaml/-/yaml-2.9.1.tgz","integrity":"sha512-3NxN8+78OdzbT7C/WjGsyfPAtJaN3FNDsWxv7Y7mcDsT/oOmgW8BpyQQFFBnvZE3j9Y2Sdz1ULFLezL7Eb2yFw=="}}}' > package-lock.json && npm ci --ignore-scripts --no-audit --no-fund && echo "CODE_GRAPH_DEPS_DIR=$RUNNER_TEMP/code-graph-deps" >> "$GITHUB_ENV" || { echo "::warning::graph dependencies failed their lockfile-enforced install; continuing without them"; rm -rf "$RUNNER_TEMP/code-graph-deps"; exit 1; }
- name: Build and upload the code graph
id: graph
env:
WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }}
CODE_GRAPH_DEPS_DIR: ${{ env.CODE_GRAPH_DEPS_DIR }}
GITHUB_REPOSITORY: ${{ github.repository }}
run: node /tmp/code-graph-build.mjs
doc-pipeline:
runs-on: ubuntu-latest
timeout-minutes: 720 # 12 hours for large repositories with many files
# Never on push (a push registers the workflow and runs the code-graph job
# only), never on the graph-only re-dispatch, never on a graph-only manual
# run. This is what lets workflow_dispatch API calls work on feature branches.
if: github.event_name != 'push' && github.event.action != 'flamingo-code-graph' && github.event.inputs.graph_only != 'true'
steps:
# =========================================================================
# REPORT CAPABILITY FIRST (shared failure-net standard with the code-review
# workflow): workflow-helpers.sh — which carries send_webhook and the
# report/stage-callback helpers — downloads in its OWN step before anything
# else, so a failure in the main script download below can still be pinged
# and reported home instead of leaving a phantom pending/running row.
# =========================================================================
- name: Download Report Helpers
env:
WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }}
run: |
# Function to download and verify script
SCRIPT_MANIFEST=/tmp/flamingo-script-manifest.json
# WEBHOOK_SECRET reaches curl through a 0600 config file, never argv — see
# curlAuthPreamble, which always traps the removal.
CURL_CFG=$(mktemp) && chmod 600 "$CURL_CFG"
trap 'rm -f "$CURL_CFG"' EXIT
printf 'header = "Authorization: Bearer %s"\n' "$WEBHOOK_SECRET" > "$CURL_CFG"
# The canonical scripts surface and the pre-rename one. load_script_manifest
# picks whichever this deployment actually serves and pins SCRIPTS_BASE_URL.
CI_SCRIPTS_URL="${HUB_BASE_URL%/}/api/ci/scripts"
LEGACY_SCRIPTS_URL="${HUB_BASE_URL%/}/api/doc-orchestrator/scripts"
# _try_manifest <base> — 0 loaded, 1 no manifest surface there, 2 fatal.
# The manifest is asked for ONE group: its keys are the files to download.
_try_manifest() {
local base="$1" code
code=$(curl -sS -w '%{http_code}' -o "$SCRIPT_MANIFEST" \
-K "$CURL_CFG" \
"$base/manifest.json?group=$SCRIPT_GROUP") || code="000"
if [ "$code" = "404" ]; then rm -f "$SCRIPT_MANIFEST"; return 1; fi
if [ "$code" != "200" ]; then
echo "❌ manifest request to $base failed (HTTP $code)"
rm -f "$SCRIPT_MANIFEST"
return 2
fi
# The digests are the TOP-LEVEL object. successResponse is the standard
# emitter but it does NOT add a wrapper — it is NextResponse.json(data)
# plus the no-store header — so there is no .data to reach through.
# A 200 that is not a manifest is how a hub which does not serve this path
# answers (the proxy rewrites unknown routes and returns HTML), so it
# means "wrong surface", not "corrupt".
if ! jq -e 'type == "object" and length > 0 and (to_entries | all(.value | type == "string"))' "$SCRIPT_MANIFEST" >/dev/null 2>&1; then
rm -f "$SCRIPT_MANIFEST"
return 1
fi
return 0
}
# load_script_manifest <group>
load_script_manifest() {
SCRIPT_GROUP="$1"
# The scripts surface was renamed from /api/doc-orchestrator/scripts to the
# pipeline-neutral /api/ci/scripts (it always served BOTH pipelines). The
# workflow file ships in the repo and the routes ship with the deployment,
# so the two are one version apart in BOTH directions across the rollout.
# Probe the canonical surface, fall back to the legacy one, and let the
# winner decide SCRIPTS_BASE_URL for every download that follows.
# "cmd; rc=$?" dies under the set -euo pipefail these steps run with —
# errexit fires before rc is read and the step ends with NO output. And
# "if ! cmd; then rc=$?" is worse: inside the branch $? is the status of
# the NEGATION (0), so every failure reads as success. "|| rc=$?" is the
# one form that both suppresses errexit and preserves the real code.
local rc=0
_try_manifest "$CI_SCRIPTS_URL" || rc=$?
if [ "$rc" = "0" ]; then
SCRIPTS_BASE_URL="$CI_SCRIPTS_URL"
echo "✅ script manifest loaded ($(jq -r 'length' "$SCRIPT_MANIFEST") scripts)"
return 0
fi
if [ "$rc" = "2" ]; then exit 1; fi
echo "::warning::this hub does not serve $CI_SCRIPTS_URL — falling back to the legacy $LEGACY_SCRIPTS_URL; it predates the rename"
SCRIPTS_BASE_URL="$LEGACY_SCRIPTS_URL"
rc=0
_try_manifest "$LEGACY_SCRIPTS_URL" || rc=$?
if [ "$rc" = "0" ]; then
echo "✅ script manifest loaded ($(jq -r 'length' "$SCRIPT_MANIFEST") scripts)"
return 0
fi
if [ "$rc" = "2" ]; then exit 1; fi
# Neither surface published a manifest: a hub older than the manifest
# itself. The manifest is the file list, so there is nothing to download.
echo "❌ no script manifest on this hub ($HUB_BASE_URL): it cannot name the $SCRIPT_GROUP scripts. Redeploy the hub."
exit 1
}
# download_script_group <group> — the hub names the files, this workflow
# names only the group. Downloads every script of the group, in served order.
download_script_group() {
load_script_manifest "$1"
local name
# The loop runs in THIS shell (no pipe), so a failed download exits the step.
while IFS= read -r name; do
download_and_verify "$name"
done < <(jq -r 'keys_unsorted[]' "$SCRIPT_MANIFEST")
}
download_and_verify() {
local script_name="$1"
local output_path="/tmp/$script_name"
local expected_hash
expected_hash=$(jq -r --arg n "$script_name" '.[$n] // empty' "$SCRIPT_MANIFEST")
if [ -z "$expected_hash" ]; then
echo "❌ $script_name is not in the server's script manifest!"
echo " The hub serves no such script, or it failed to read on the server."
exit 1
fi
if ! printf '%s' "$expected_hash" | grep -Eq '^[0-9a-f]{64}$'; then
echo "❌ the manifest entry for $script_name is not a SHA-256 digest — refusing to run it."
exit 1
fi
curl -fsSL "$SCRIPTS_BASE_URL/$script_name" \
-K "$CURL_CFG" \
-o "$output_path"
local actual_hash=$(shasum -a 256 "$output_path" | cut -d' ' -f1)
if [ "$actual_hash" != "$expected_hash" ]; then
echo "❌ HASH MISMATCH for $script_name!"
echo " Expected: $expected_hash"
echo " Actual: $actual_hash"
echo " The download was corrupted in transit — both values come from the same deployment."
exit 1
fi
# Make shell scripts executable
if [[ "$script_name" == *.sh ]]; then
chmod +x "$output_path"
fi
echo "✅ $script_name verified (hash: ${actual_hash:0:16}...)"
}
# Digests AND the file list come from the deployment serving the bytes, not from this file.
download_script_group "doc-helpers"
# =========================================================================
# SEND START NOTIFICATION — the early "the workflow actually started" ping.
# Deliberately BEFORE the main script download: it stamps workflow_run_id +
# status 'running' on the hub's run row, which is what lets the hub's tiered
# reaper tell "dispatch accepted but nothing ran" (never pinged — failed
# fast) from "started and then crashed" (pinged — longer deadline).
# =========================================================================
- name: Send Start Notification
if: env.HUB_BASE_URL != ''
continue-on-error: true
env:
WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }}
run: |
source /tmp/workflow-helpers.sh
# Construct callback URL from hub base URL
CALLBACK_URL="${HUB_BASE_URL}/api/doc-orchestrator/webhook"
echo "📤 Sending start notification to: $CALLBACK_URL"
PAYLOAD="{
\"run_id\": \"$RUN_ID\",
\"repo_id\": \"$REPO_ID\",
\"status\": \"running\",
\"workflow_run_id\": ${{ github.run_id }},
\"workflow_url\": \"${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}\",
\"current_stage\": \"inline-docs\"
}"
HTTP_CODE=$(send_webhook "$CALLBACK_URL" "$WEBHOOK_SECRET" "$PAYLOAD" "/tmp/webhook_start_response.txt") || HTTP_CODE="failed"
if [ "$HTTP_CODE" = "200" ] || [ "$HTTP_CODE" = "201" ]; then
echo "✅ Start notification sent (HTTP $HTTP_CODE)"
else
echo "⚠️ Start notification returned HTTP $HTTP_CODE (non-blocking)"
fi
# =========================================================================
# DOWNLOAD WORKFLOW SCRIPTS
# Downloads all reusable scripts from authenticated admin-hub endpoint
# (workflow-helpers.sh already downloaded by the report-capability step)
# =========================================================================
- name: Download Workflow Scripts
id: helpers
env:
WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }}
# No HASH_* pins: the digests come from manifest.json on the same
# endpoint that serves the scripts, so a hash in this file can never
# be a different ref's than the bytes it checks.
run: |
echo "📥 Downloading workflow scripts..."
# Function to download and verify script
SCRIPT_MANIFEST=/tmp/flamingo-script-manifest.json
# WEBHOOK_SECRET reaches curl through a 0600 config file, never argv — see
# curlAuthPreamble, which always traps the removal.
CURL_CFG=$(mktemp) && chmod 600 "$CURL_CFG"
trap 'rm -f "$CURL_CFG"' EXIT
printf 'header = "Authorization: Bearer %s"\n' "$WEBHOOK_SECRET" > "$CURL_CFG"
# The canonical scripts surface and the pre-rename one. load_script_manifest
# picks whichever this deployment actually serves and pins SCRIPTS_BASE_URL.
CI_SCRIPTS_URL="${HUB_BASE_URL%/}/api/ci/scripts"
LEGACY_SCRIPTS_URL="${HUB_BASE_URL%/}/api/doc-orchestrator/scripts"
# _try_manifest <base> — 0 loaded, 1 no manifest surface there, 2 fatal.
# The manifest is asked for ONE group: its keys are the files to download.
_try_manifest() {
local base="$1" code
code=$(curl -sS -w '%{http_code}' -o "$SCRIPT_MANIFEST" \
-K "$CURL_CFG" \
"$base/manifest.json?group=$SCRIPT_GROUP") || code="000"
if [ "$code" = "404" ]; then rm -f "$SCRIPT_MANIFEST"; return 1; fi
if [ "$code" != "200" ]; then
echo "❌ manifest request to $base failed (HTTP $code)"
rm -f "$SCRIPT_MANIFEST"
return 2
fi
# The digests are the TOP-LEVEL object. successResponse is the standard
# emitter but it does NOT add a wrapper — it is NextResponse.json(data)
# plus the no-store header — so there is no .data to reach through.
# A 200 that is not a manifest is how a hub which does not serve this path
# answers (the proxy rewrites unknown routes and returns HTML), so it
# means "wrong surface", not "corrupt".
if ! jq -e 'type == "object" and length > 0 and (to_entries | all(.value | type == "string"))' "$SCRIPT_MANIFEST" >/dev/null 2>&1; then
rm -f "$SCRIPT_MANIFEST"
return 1
fi
return 0
}
# load_script_manifest <group>
load_script_manifest() {
SCRIPT_GROUP="$1"
# The scripts surface was renamed from /api/doc-orchestrator/scripts to the
# pipeline-neutral /api/ci/scripts (it always served BOTH pipelines). The
# workflow file ships in the repo and the routes ship with the deployment,
# so the two are one version apart in BOTH directions across the rollout.
# Probe the canonical surface, fall back to the legacy one, and let the
# winner decide SCRIPTS_BASE_URL for every download that follows.
# "cmd; rc=$?" dies under the set -euo pipefail these steps run with —
# errexit fires before rc is read and the step ends with NO output. And
# "if ! cmd; then rc=$?" is worse: inside the branch $? is the status of
# the NEGATION (0), so every failure reads as success. "|| rc=$?" is the
# one form that both suppresses errexit and preserves the real code.
local rc=0
_try_manifest "$CI_SCRIPTS_URL" || rc=$?
if [ "$rc" = "0" ]; then
SCRIPTS_BASE_URL="$CI_SCRIPTS_URL"
echo "✅ script manifest loaded ($(jq -r 'length' "$SCRIPT_MANIFEST") scripts)"
return 0
fi
if [ "$rc" = "2" ]; then exit 1; fi
echo "::warning::this hub does not serve $CI_SCRIPTS_URL — falling back to the legacy $LEGACY_SCRIPTS_URL; it predates the rename"
SCRIPTS_BASE_URL="$LEGACY_SCRIPTS_URL"
rc=0
_try_manifest "$LEGACY_SCRIPTS_URL" || rc=$?
if [ "$rc" = "0" ]; then
echo "✅ script manifest loaded ($(jq -r 'length' "$SCRIPT_MANIFEST") scripts)"
return 0
fi
if [ "$rc" = "2" ]; then exit 1; fi
# Neither surface published a manifest: a hub older than the manifest
# itself. The manifest is the file list, so there is nothing to download.
echo "❌ no script manifest on this hub ($HUB_BASE_URL): it cannot name the $SCRIPT_GROUP scripts. Redeploy the hub."
exit 1
}
# download_script_group <group> — the hub names the files, this workflow
# names only the group. Downloads every script of the group, in served order.
download_script_group() {
load_script_manifest "$1"
local name
# The loop runs in THIS shell (no pipe), so a failed download exits the step.
while IFS= read -r name; do
download_and_verify "$name"
done < <(jq -r 'keys_unsorted[]' "$SCRIPT_MANIFEST")
}
download_and_verify() {
local script_name="$1"
local output_path="/tmp/$script_name"
local expected_hash
expected_hash=$(jq -r --arg n "$script_name" '.[$n] // empty' "$SCRIPT_MANIFEST")
if [ -z "$expected_hash" ]; then
echo "❌ $script_name is not in the server's script manifest!"
echo " The hub serves no such script, or it failed to read on the server."
exit 1
fi
if ! printf '%s' "$expected_hash" | grep -Eq '^[0-9a-f]{64}$'; then
echo "❌ the manifest entry for $script_name is not a SHA-256 digest — refusing to run it."
exit 1
fi
curl -fsSL "$SCRIPTS_BASE_URL/$script_name" \
-K "$CURL_CFG" \
-o "$output_path"
local actual_hash=$(shasum -a 256 "$output_path" | cut -d' ' -f1)
if [ "$actual_hash" != "$expected_hash" ]; then
echo "❌ HASH MISMATCH for $script_name!"
echo " Expected: $expected_hash"
echo " Actual: $actual_hash"
echo " The download was corrupted in transit — both values come from the same deployment."
exit 1
fi
# Make shell scripts executable
if [[ "$script_name" == *.sh ]]; then
chmod +x "$output_path"
fi
echo "✅ $script_name verified (hash: ${actual_hash:0:16}...)"
}
# Every script a documentation run uses, stage 0 (the code graph) included.
# Digests AND the file list come from the deployment serving the bytes,
# not from this file: the hub lists the group in download order (a helper
# a generator require()s at load comes before it).
download_script_group "doc-pipeline"
# ONE vocabulary read for the whole run: every later step (language
# detection, source discovery, the generators, the graph build) reads this
# file, so none of them needs the secret for it.
node /tmp/ci-source.mjs vocabulary /tmp/ci-vocabulary.json || { echo "❌ could not read the source vocabulary from the hub"; exit 1; }
echo "CODE_GRAPH_VOCABULARY_FILE=/tmp/ci-vocabulary.json" >> $GITHUB_ENV
echo "📦 All 13 workflow files downloaded and verified (12 scripts + 1 ruleset; helpers verified earlier)"
# Export paths for all stages (use os.tmpdir() compatible paths)
echo "VALIDATION_RULES_PATH=/tmp/markdown-validation-rules.md" >> $GITHUB_ENV
echo "GUIDELINES_PATH=/tmp/flamingo-markdown-guidelines.md" >> $GITHUB_ENV
echo "STAGE3_FILES_TRACKER=/tmp/stage3-files.txt" >> $GITHUB_ENV
echo "STAGE3_STATS_FILE=/tmp/.doc-stage3-stats.json" >> $GITHUB_ENV
echo "STAGE4_FILES_TRACKER=/tmp/stage4-files.txt" >> $GITHUB_ENV
# Download Flamingo Markdown Guidelines (separate endpoint)
# REQUIRED: Guidelines are needed for markdown validation and CodeWiki prompts
echo ""
echo "📋 Downloading Flamingo Markdown Guidelines..."
GUIDELINES_URL="${HUB_BASE_URL}/api/doc-orchestrator/guidelines"
# Same 0600 config file the download block above set up.
HTTP_CODE=$(curl -fsSL -w "%{http_code}" \
"$GUIDELINES_URL" \
-K "$CURL_CFG" \
-o "/tmp/flamingo-markdown-guidelines.md" 2>/dev/null) || HTTP_CODE="failed"
if [ "$HTTP_CODE" = "200" ]; then
GUIDELINES_SIZE=$(wc -c < /tmp/flamingo-markdown-guidelines.md | tr -d ' ')
if [ "$GUIDELINES_SIZE" -lt 100 ]; then
echo "❌ Guidelines file too small ($GUIDELINES_SIZE bytes) - likely an error response"
cat /tmp/flamingo-markdown-guidelines.md
exit 1
fi
echo "✅ Guidelines downloaded ($GUIDELINES_SIZE bytes)"
else
echo "❌ Guidelines download failed (HTTP $HTTP_CODE)"
echo " URL: $GUIDELINES_URL"
echo " Guidelines are required for markdown validation and CodeWiki prompts."
echo " Check that the guidelines endpoint is deployed and working."
rm -f /tmp/flamingo-markdown-guidelines.md
exit 1
fi
# (Send Start Notification moved ABOVE the main script download — see the
# report-capability step ordering at the top of the job.)
- name: Checkout Repository
# v5 = the Node 24 drop-in (v4 targets EOL Node 20 and warns on every run).
uses: actions/checkout@v5
with:
fetch-depth: 0
token: ${{ secrets.GITHUB_TOKEN }}
ref: ${{ env.SOURCE_BRANCH }}
# The SOURCE head, before the PR branch and the Clean Slate commit move
# HEAD: the stage-0 graph build tags this commit (the code being
# documented), never the docs branch it is sitting on.
- name: Record source head
run: echo "SOURCE_HEAD_SHA=$(git rev-parse HEAD)" >> "$GITHUB_ENV"
# =========================================================================
# SETUP: Clone Dependency Repos (if configured)
# =========================================================================
- name: Clone Dependency Repositories
if: env.DEPENDENCIES != ''
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
# SECURITY: Pass secret per-step with inline masking
GITHUB_PAT: ${{ secrets.DOC_ORCH_GITHUB_PAT }}
run: |
source /tmp/workflow-helpers.sh
echo "📦 Cloning dependency repositories for documentation context..."
echo " Dependencies: $DEPENDENCIES"
ensure_directory "../deps"
# Determine which token to use (PAT preferred for cross-repo access)
if [ -n "$GITHUB_PAT" ]; then
echo " Using provided GitHub PAT for cross-repo access"
CLONE_TOKEN="$GITHUB_PAT"
else
echo " Using default GITHUB_TOKEN (may not work for private cross-repo)"
CLONE_TOKEN="$GH_TOKEN"
fi
# Configure git to use token for private repos
git config --global url."https://x-access-token:${CLONE_TOKEN}@github.com/".insteadOf "https://github.com/"
IFS=',' read -ra DEPS <<< "$DEPENDENCIES"
for dep in "${DEPS[@]}"; do
repo_name=$(basename $dep)
echo ""
echo " 📁 Cloning: $dep → ../deps/$repo_name"
if git clone --depth 1 "https://github.com/$dep.git" "../deps/$repo_name" 2>&1; then
file_count=$(node /tmp/ci-source.mjs count "../deps/$repo_name" 2>/dev/null || echo "?")
echo " ✅ Cloned successfully ($file_count source files)"
else
echo " ⚠️ Failed to clone $dep"
echo " If private, pass github_pat with 'repo' scope"
fi
done
echo ""
echo "📂 Dependency directories:"
ls -la ../deps/ 2>/dev/null || echo " No dependencies cloned"
echo ""
echo "📊 Total dependency source files available for documentation:"
echo " $(node /tmp/ci-source.mjs count ../deps 2>/dev/null || echo '?')"
# =========================================================================
# LANGUAGE DETECTION (runs before all stages for consistency)
# Determines primary language for filtering in Stage 1, 2, and 3
# =========================================================================
- name: Detect Repository Language
id: detect_language
run: |
source /tmp/workflow-helpers.sh
# This step is the FIRST reader of CODEWIKI_CONFIG_JSON — it runs before
# "Validate and Parse Input Parameters" — so the emptiness guard lives here,
# ahead of the first jq, rather than in the later validation step.
if [ -z "$CODEWIKI_CONFIG_JSON" ]; then
echo "❌ Missing: CODEWIKI_CONFIG_JSON (required before language detection)"
exit 1
fi
echo "🔍 Detecting repository primary language..."
echo " Scanning main repo (.) and dependency repos (../deps/)"
# ONE detection, from the vocabulary the hub serves (ci-source.mjs): which
# languages are source, their extensions, and which of them CodeWiki can
# parse are rows in the hub's language table. This step used to carry nine
# hand-typed `find` counts, a positional helper and a six-way threshold
# test, each with its own idea of the extensions and the exclusions.
DETECTION=$(node /tmp/ci-source.mjs detect) || { echo "❌ language detection failed"; exit 1; }
PRIMARY_LANG=$(echo "$DETECTION" | jq -r '.primary')
MAX_COUNT=$(echo "$DETECTION" | jq -r '.max')
CODEWIKI_SUPPORTED=$(echo "$DETECTION" | jq -r '.codewiki_supported')
echo ""
echo "📊 Source files by language (tests and never-source directories excluded):"
echo "$DETECTION" | jq -r '.counts | to_entries[] | select(.value > 0) | " \(.key): \(.value)"'
echo ""
echo "✅ Primary language: $PRIMARY_LANG ($MAX_COUNT files)"
if [ "$CODEWIKI_SUPPORTED" = "true" ]; then
echo " CodeWiki supported: YES"
else
echo " CodeWiki supported: NO (will use Claude Architecture Analysis)"
fi
# Per-repo engine override.
#
# The detection above cannot see mixed repos: it counts `.` AND `../deps`, so
# a Rust or Go product with a TypeScript dependency clones its way past the
# >=10 threshold and runs CodeWiki over a codebase whose analyzers do not
# exist — which yields synthetic module_1/module_2/... docs that look like a
# successful run. `engine` pins the choice.
#
# Applied here, before set_output, so all five downstream gates keep reading
# one value and need no change. It cannot live in the `if:` conditions:
# GitHub Actions expressions have no ternary.
CODEWIKI_ENGINE=$(require_json_key "$CODEWIKI_CONFIG_JSON" '.engine' 'codewiki engine') || exit 1
case "$CODEWIKI_ENGINE" in
claude)
CODEWIKI_SUPPORTED="false"
echo " ⚙️ engine=claude — forcing Claude Architecture Analysis"
;;
codewiki)
CODEWIKI_SUPPORTED="true"
echo " ⚙️ engine=codewiki — forcing CodeWiki"
;;
auto)
echo " ⚙️ engine=auto — using detected value ($CODEWIKI_SUPPORTED)"
;;
*)
echo "❌ Invalid engine '$CODEWIKI_ENGINE' (expected auto|codewiki|claude)"
exit 1
;;
esac
# Output for use by subsequent steps
set_output "primary_language" "$PRIMARY_LANG"
set_output "codewiki_supported" "$CODEWIKI_SUPPORTED"
set_output "file_count" "$MAX_COUNT"
# =========================================================================
# VALIDATE AND PARSE ALL INPUT PARAMETERS
# 1. Validate all required parameters are present
# 2. Parse ALL JSON configurations into individual environment variables
# 3. Validate parsed values
# Required to stay under GitHub Actions 25-parameter limit
# =========================================================================
- name: Validate and Parse Input Parameters
run: |
echo "🔍 Validating and parsing all input parameters..."
echo ""
# ===================================================================
# 1. VALIDATE REQUIRED PARAMETERS
# ===================================================================
echo "1️⃣ Validating required parameters..."
VALIDATION_FAILED=0
# Core parameters
[ -z "$RUN_ID" ] && echo "❌ Missing: RUN_ID" && VALIDATION_FAILED=1
[ -z "$REPO_ID" ] && echo "❌ Missing: REPO_ID" && VALIDATION_FAILED=1
[ -z "$HUB_BASE_URL" ] && echo "❌ Missing: HUB_BASE_URL" && VALIDATION_FAILED=1
[ -z "$STAGES" ] && echo "❌ Missing: STAGES" && VALIDATION_FAILED=1
[ -z "$CLAUDE_MODEL" ] && echo "❌ Missing: CLAUDE_MODEL" && VALIDATION_FAILED=1
# JSON parameters
[ -z "$CODEWIKI_CONFIG_JSON" ] && echo "❌ Missing: CODEWIKI_CONFIG_JSON" && VALIDATION_FAILED=1
[ -z "$OUTPUT_PATHS_JSON" ] && echo "❌ Missing: OUTPUT_PATHS_JSON" && VALIDATION_FAILED=1
[ -z "$README_CONFIG_JSON" ] && echo "❌ Missing: README_CONFIG_JSON" && VALIDATION_FAILED=1
[ -z "$YOUTUBE_CONFIG_JSON" ] && echo "❌ Missing: YOUTUBE_CONFIG_JSON" && VALIDATION_FAILED=1
if [ $VALIDATION_FAILED -eq 1 ]; then
echo ""
echo "❌ Validation failed: Missing required parameters"
exit 1
fi
echo " ✅ All required parameters present"
echo ""
# ===================================================================
# 2. PARSE CODEWIKI CONFIGURATION (nested JSONB structure)
# ===================================================================
echo "2️⃣ Parsing CodeWiki configuration..."
# Only the keys with a real consumer are extracted here — the per-phase
# base_url / api_version / temperature / temperature_supported are read
# directly from CODEWIKI_CONFIG_JSON by configure_codewiki_from_json, which
# is the single place that builds the `codewiki config set` command. They
# used to be parsed here as well and exported to $GITHUB_ENV, where nothing
# read them.
#
# No `// default` fallbacks anywhere below. The hub deep-merges every JSON
# param against the params SSOT before dispatch, so an absent key is a real
# bug — and a fallback here would silently win over the SSOT, which is how
# docs/architecture and docs/reference/architecture drifted apart.
# `require_json_key` / `optional_json_key` come from workflow-helpers.sh.
source /tmp/workflow-helpers.sh
CW_JSON="$CODEWIKI_CONFIG_JSON"
# Parse nested cluster config
CODEWIKI_CLUSTER_PROVIDER=$(require_json_key "$CW_JSON" '.cluster.provider' 'cluster provider') || exit 1
CODEWIKI_CLUSTER_MODEL=$(require_json_key "$CW_JSON" '.cluster.model' 'cluster model') || exit 1
CODEWIKI_CLUSTER_MAX_TOKENS=$(require_json_key "$CW_JSON" '.cluster.max_tokens' 'cluster max_tokens') || exit 1
CODEWIKI_CLUSTER_MAX_TOKEN_FIELD=$(require_json_key "$CW_JSON" '.cluster.max_token_field' 'cluster max_token_field') || exit 1
# Nullable by design: api_version is null for every OpenAI model.
# Parse nested generation config
CODEWIKI_GENERATION_PROVIDER=$(require_json_key "$CW_JSON" '.generation.provider' 'generation provider') || exit 1
CODEWIKI_GENERATION_MODEL=$(require_json_key "$CW_JSON" '.generation.model' 'generation model') || exit 1
CODEWIKI_GENERATION_MAX_TOKENS=$(require_json_key "$CW_JSON" '.generation.max_tokens' 'generation max_tokens') || exit 1
CODEWIKI_GENERATION_MAX_TOKEN_FIELD=$(require_json_key "$CW_JSON" '.generation.max_token_field' 'generation max_token_field') || exit 1
# Parse nested fallback config
CODEWIKI_FALLBACK_PROVIDER=$(require_json_key "$CW_JSON" '.fallback.provider' 'fallback provider') || exit 1
CODEWIKI_FALLBACK_MODEL=$(require_json_key "$CW_JSON" '.fallback.model' 'fallback model') || exit 1
CODEWIKI_FALLBACK_MAX_TOKENS=$(require_json_key "$CW_JSON" '.fallback.max_tokens' 'fallback max_tokens') || exit 1
CODEWIKI_FALLBACK_MAX_TOKEN_FIELD=$(require_json_key "$CW_JSON" '.fallback.max_token_field' 'fallback max_token_field') || exit 1
# Parse top-level config.
# max_files_per_module is LIVE: this value reaches CodeWiki through the
# job-scoped $GITHUB_ENV write below, and upstream reads it in its
# empty-module-tree branch — the branch Go/Rust/HCL repos land in.
CODEWIKI_MAX_FILES_PER_MODULE=$(require_json_key "$CW_JSON" '.max_files_per_module' 'max_files_per_module') || exit 1
CODEWIKI_MAX_DEPTH=$(require_json_key "$CW_JSON" '.max_depth' 'max_depth') || exit 1
CODEWIKI_REPO=$(require_json_key "$CW_JSON" '.repo' 'codewiki repo url') || exit 1
echo " Cluster (Phase 2): $CODEWIKI_CLUSTER_PROVIDER / $CODEWIKI_CLUSTER_MODEL (${CODEWIKI_CLUSTER_MAX_TOKEN_FIELD}, ${CODEWIKI_CLUSTER_MAX_TOKENS} tokens)"
echo " Generation (Phase 3+): $CODEWIKI_GENERATION_PROVIDER / $CODEWIKI_GENERATION_MODEL (${CODEWIKI_GENERATION_MAX_TOKEN_FIELD}, ${CODEWIKI_GENERATION_MAX_TOKENS} tokens)"
echo " Fallback: $CODEWIKI_FALLBACK_PROVIDER / $CODEWIKI_FALLBACK_MODEL (${CODEWIKI_FALLBACK_MAX_TOKEN_FIELD}, ${CODEWIKI_FALLBACK_MAX_TOKENS} tokens)"
echo " Max depth: $CODEWIKI_MAX_DEPTH"
echo " Max files/module: $CODEWIKI_MAX_FILES_PER_MODULE"
echo " ✅ CodeWiki parameters parsed"
echo ""
# ===================================================================
# 3. PARSE YOUTUBE CONFIGURATION (JSONB structure)
# ===================================================================
echo "3️⃣ Parsing YouTube configuration..."
# Parse YouTube config from JSONB (channels only - API key from secrets)
# channels is legitimately optional: no channels == feature off
YOUTUBE_CHANNELS=$(echo "$YOUTUBE_CONFIG_JSON" | jq -c '.channels // []')
# YouTube is enabled if channels array has items
YOUTUBE_ENABLED=$(echo "$YOUTUBE_CHANNELS" | jq -r 'if length > 0 then "true" else "false" end')
echo " Enabled: $YOUTUBE_ENABLED (based on channels count)"
echo " Channels: $YOUTUBE_CHANNELS"
echo " API key source: secrets.YOUTUBE_API_KEY (SECURE - not in database)"
echo " ✅ YouTube configuration parsed"
echo ""
# ===================================================================
# 4. PARSE README CONFIGURATION (JSONB structure)
# ===================================================================
echo "4️⃣ Parsing README logo configuration..."
# Parse README config from JSONB
README_LOGO_DARK=$(optional_json_key "$README_CONFIG_JSON" '.logo_dark')
README_LOGO_LIGHT=$(optional_json_key "$README_CONFIG_JSON" '.logo_light')
README_LOGO_ALT=$(require_json_key "$README_CONFIG_JSON" '.logo_alt' 'readme logo alt') || exit 1
echo " Dark logo: ${README_LOGO_DARK:-'(not set)'}"
echo " Light logo: ${README_LOGO_LIGHT:-'(not set)'}"
echo " Alt text: $README_LOGO_ALT"
echo " ✅ README logo configuration parsed"
echo ""
# ===================================================================
# 5. PARSE OUTPUT PATHS CONFIGURATION (5 params)
# ===================================================================
echo "5️⃣ Parsing output paths..."
# Same fail-loud contract as the codewiki block. These fallbacks were the
# last surviving copy of the five paths, and `.reference` still said
# docs/architecture while the SSOT said docs/reference/architecture — the
# very drift the SSOT was created to end.
OP_JSON="$OUTPUT_PATHS_JSON"
DOCS_OUTPUT_PATH=$(require_json_key "$OP_JSON" '.docs' 'docs output path') || exit 1
REFERENCE_OUTPUT_PATH=$(require_json_key "$OP_JSON" '.reference' 'reference output path') || exit 1
DIAGRAMS_OUTPUT_PATH=$(require_json_key "$OP_JSON" '.diagrams' 'diagrams output path') || exit 1
GETTING_STARTED_OUTPUT_PATH=$(require_json_key "$OP_JSON" '.getting_started' 'getting-started output path') || exit 1
DEVELOPMENT_OUTPUT_PATH=$(require_json_key "$OP_JSON" '.development' 'development output path') || exit 1
echo " Base docs: $DOCS_OUTPUT_PATH"
echo " Reference: $REFERENCE_OUTPUT_PATH"
echo " Diagrams: $DIAGRAMS_OUTPUT_PATH"
echo " Getting started: $GETTING_STARTED_OUTPUT_PATH"
echo " Development: $DEVELOPMENT_OUTPUT_PATH"
echo " ✅ 5 output path parameters parsed"
echo ""
# ===================================================================
# 6. SET CUSTOM INSTRUCTIONS & EXTERNAL REPOS (separate parameters)
# ===================================================================
echo "6️⃣ Setting custom instructions and external repos..."
# ===================================================================
# 6a. BUILD REPOSITORY CONTEXT (prevents AI URL hallucinations)
# ===================================================================
# Extract repository information from GitHub context
GITHUB_REPO="${{ github.repository }}"
GITHUB_OWNER="${{ github.repository_owner }}"
GITHUB_SERVER="${{ github.server_url }}"
GITHUB_REPO_NAME=$(echo "$GITHUB_REPO" | cut -d'/' -f2)
GITHUB_REPO_URL="${GITHUB_SERVER}/${GITHUB_REPO}"
# Build repository context section (injected into ALL AI prompts)
# Use printf for multi-line string (avoids YAML parsing issues with heredoc)
printf -v REPOSITORY_CONTEXT '%s\n' \
'## REPOSITORY CONTEXT - GROUND TRUTH' \
'' \
'**CRITICAL:** This section provides the ACTUAL repository information. You MUST use these exact values when constructing GitHub URLs.' \
'' \
"- **Repository:** ${GITHUB_REPO}" \
"- **Owner:** ${GITHUB_OWNER}" \
"- **Repository Name:** ${GITHUB_REPO_NAME}" \
"- **Repository URL:** ${GITHUB_REPO_URL}" \
"- **Server:** ${GITHUB_SERVER}" \
'' \
'**MANDATORY RULES FOR GITHUB URLS:**' \
"1. ALWAYS use the exact repository path: \`${GITHUB_REPO}\`" \
'2. NEVER use placeholder URLs like "your-org", "example-org", or "mycompany"' \
'3. NEVER infer repository owner from file contents or dependencies' \
'4. NEVER use upstream/parent repository URLs (if this is a fork, use the fork URL)' \
"5. When linking to code: \`${GITHUB_REPO_URL}/blob/main/path/to/file\`" \
"6. When linking to clone: \`git clone ${GITHUB_REPO_URL}.git\`" \
"7. When linking to issues/PRs: \`${GITHUB_REPO_URL}/issues\` or \`${GITHUB_REPO_URL}/pulls\`" \
"8. When linking to releases: \`${GITHUB_REPO_URL}/releases\`" \
'' \
'**If you find yourself writing a GitHub URL, verify it matches the Repository URL above.**' \
"**ESPECIALLY IN README.md and tutorials - All GitHub URLs MUST use ${GITHUB_REPO}**" \
'' \
'---'
echo " Repository: $GITHUB_REPO"
echo " Repository URL: $GITHUB_REPO_URL"
echo " Repository context length: ${#REPOSITORY_CONTEXT} chars"
# DEBUG: Print first 200 chars of repository context to verify it's set
echo " Repository context preview: ${REPOSITORY_CONTEXT:0:200}..."
# ===================================================================
# 6b. PREPEND REPOSITORY CONTEXT TO CUSTOM INSTRUCTIONS
# ===================================================================
# Custom instructions come as plain text from user
# Read from the environment rather than interpolating into single quotes.
# GitHub Actions expression substitution runs BEFORE bash parses the line, so a
# single apostrophe anywhere in an admin's instructions used to terminate the
# string and kill the step with a syntax error. NOTE: never write a literal
# empty GitHub expression in this run block, even inside a comment — Actions
# evaluates it pre-bash and the whole workflow fails to parse.
USER_CUSTOM_INSTRUCTIONS="$CUSTOM_REPO_INSTRUCTIONS"
# Combine repository context + user custom instructions
# Repository context goes FIRST (highest priority in prompts)
if [ -n "$USER_CUSTOM_INSTRUCTIONS" ]; then
CUSTOM_INSTRUCTIONS="${REPOSITORY_CONTEXT}"$'\n\n'"${USER_CUSTOM_INSTRUCTIONS}"
else
CUSTOM_INSTRUCTIONS="$REPOSITORY_CONTEXT"
fi
# External repos come as separate JSON array parameter
EXTERNAL_REPOS_COUNT=$(echo "$EXTERNAL_REPOS" | jq '. | length' 2>/dev/null || echo "0")
echo " User instructions length: ${#USER_CUSTOM_INSTRUCTIONS} chars"
echo " Repository context length: ${#REPOSITORY_CONTEXT} chars"
echo " Total instructions length (with repo context): ${#CUSTOM_INSTRUCTIONS} chars"
echo " External repos: $EXTERNAL_REPOS_COUNT repos"
# DEBUG: Print first 300 chars of CUSTOM_INSTRUCTIONS to verify repository context is included
echo " ===== CUSTOM_INSTRUCTIONS PREVIEW ====="
echo "${CUSTOM_INSTRUCTIONS:0:300}..."
echo " ========================================"
echo " ✅ Repository context + custom instructions + external repos set"
echo ""
# ===================================================================
# 5. VALIDATE PARSED VALUES
# ===================================================================
echo "5️⃣ Validating parsed values..."
# Validate providers
if [[ ! "$CODEWIKI_CLUSTER_PROVIDER" =~ ^(anthropic|openai)$ ]]; then
echo "❌ Invalid cluster provider: $CODEWIKI_CLUSTER_PROVIDER"
exit 1
fi
if [[ ! "$CODEWIKI_GENERATION_PROVIDER" =~ ^(anthropic|openai)$ ]]; then
echo "❌ Invalid generation provider: $CODEWIKI_GENERATION_PROVIDER"
exit 1
fi
if [[ ! "$CODEWIKI_FALLBACK_PROVIDER" =~ ^(anthropic|openai)$ ]]; then
echo "❌ Invalid fallback provider: $CODEWIKI_FALLBACK_PROVIDER"
exit 1
fi
# Validate model names are not empty
[ -z "$CODEWIKI_CLUSTER_MODEL" ] && echo "❌ Empty cluster model" && exit 1
[ -z "$CODEWIKI_GENERATION_MODEL" ] && echo "❌ Empty generation model" && exit 1
[ -z "$CODEWIKI_FALLBACK_MODEL" ] && echo "❌ Empty fallback model" && exit 1
# Validate numeric values
[[ ! "$CODEWIKI_CLUSTER_MAX_TOKENS" =~ ^[0-9]+$ ]] && echo "❌ Invalid cluster max_tokens: $CODEWIKI_CLUSTER_MAX_TOKENS" && exit 1
[[ ! "$CODEWIKI_GENERATION_MAX_TOKENS" =~ ^[0-9]+$ ]] && echo "❌ Invalid generation max_tokens: $CODEWIKI_GENERATION_MAX_TOKENS" && exit 1
[[ ! "$CODEWIKI_FALLBACK_MAX_TOKENS" =~ ^[0-9]+$ ]] && echo "❌ Invalid fallback max_tokens: $CODEWIKI_FALLBACK_MAX_TOKENS" && exit 1
[[ ! "$CODEWIKI_MAX_DEPTH" =~ ^[0-9]+$ ]] && echo "❌ Invalid max_depth: $CODEWIKI_MAX_DEPTH" && exit 1
[[ ! "$CODEWIKI_MAX_FILES_PER_MODULE" =~ ^[0-9]+$ ]] && echo "❌ Invalid max_files_per_module: $CODEWIKI_MAX_FILES_PER_MODULE" && exit 1
echo " ✅ All parsed values are valid"
echo ""
# ===================================================================
# 7. EXPORT TO GITHUB_ENV (makes values available to all steps)
# ===================================================================
echo "7️⃣ Exporting to GITHUB_ENV..."
# CodeWiki config (31 vars: cluster=9, generation=10, fallback=9, shared=3)
echo "CODEWIKI_CLUSTER_PROVIDER=$CODEWIKI_CLUSTER_PROVIDER" >> $GITHUB_ENV
echo "CODEWIKI_CLUSTER_MODEL=$CODEWIKI_CLUSTER_MODEL" >> $GITHUB_ENV
echo "CODEWIKI_CLUSTER_MAX_TOKENS=$CODEWIKI_CLUSTER_MAX_TOKENS" >> $GITHUB_ENV
echo "CODEWIKI_CLUSTER_MAX_TOKEN_FIELD=$CODEWIKI_CLUSTER_MAX_TOKEN_FIELD" >> $GITHUB_ENV
echo "CODEWIKI_GENERATION_PROVIDER=$CODEWIKI_GENERATION_PROVIDER" >> $GITHUB_ENV
echo "CODEWIKI_GENERATION_MODEL=$CODEWIKI_GENERATION_MODEL" >> $GITHUB_ENV
echo "CODEWIKI_GENERATION_MAX_TOKENS=$CODEWIKI_GENERATION_MAX_TOKENS" >> $GITHUB_ENV
echo "CODEWIKI_GENERATION_MAX_TOKEN_FIELD=$CODEWIKI_GENERATION_MAX_TOKEN_FIELD" >> $GITHUB_ENV
echo "CODEWIKI_FALLBACK_PROVIDER=$CODEWIKI_FALLBACK_PROVIDER" >> $GITHUB_ENV
echo "CODEWIKI_FALLBACK_MODEL=$CODEWIKI_FALLBACK_MODEL" >> $GITHUB_ENV
echo "CODEWIKI_FALLBACK_MAX_TOKENS=$CODEWIKI_FALLBACK_MAX_TOKENS" >> $GITHUB_ENV
echo "CODEWIKI_FALLBACK_MAX_TOKEN_FIELD=$CODEWIKI_FALLBACK_MAX_TOKEN_FIELD" >> $GITHUB_ENV
echo "CODEWIKI_MAX_FILES_PER_MODULE=$CODEWIKI_MAX_FILES_PER_MODULE" >> $GITHUB_ENV
echo "CODEWIKI_MAX_DEPTH=$CODEWIKI_MAX_DEPTH" >> $GITHUB_ENV
echo "CODEWIKI_REPO=$CODEWIKI_REPO" >> $GITHUB_ENV
# YouTube config (2 vars - API key from secrets, not exported here)
echo "YOUTUBE_ENABLED=$YOUTUBE_ENABLED" >> $GITHUB_ENV
echo "YOUTUBE_CHANNELS=$YOUTUBE_CHANNELS" >> $GITHUB_ENV
# README config (3 vars)
echo "README_LOGO_DARK=$README_LOGO_DARK" >> $GITHUB_ENV
echo "README_LOGO_LIGHT=$README_LOGO_LIGHT" >> $GITHUB_ENV
echo "README_LOGO_ALT=$README_LOGO_ALT" >> $GITHUB_ENV
# Output paths (5 vars)
echo "DOCS_OUTPUT_PATH=$DOCS_OUTPUT_PATH" >> $GITHUB_ENV
echo "REFERENCE_OUTPUT_PATH=$REFERENCE_OUTPUT_PATH" >> $GITHUB_ENV
echo "DIAGRAMS_OUTPUT_PATH=$DIAGRAMS_OUTPUT_PATH" >> $GITHUB_ENV
echo "GETTING_STARTED_OUTPUT_PATH=$GETTING_STARTED_OUTPUT_PATH" >> $GITHUB_ENV
echo "DEVELOPMENT_OUTPUT_PATH=$DEVELOPMENT_OUTPUT_PATH" >> $GITHUB_ENV
# Custom instructions (2 vars)
echo "CUSTOM_INSTRUCTIONS<<EOF" >> $GITHUB_ENV
echo "$CUSTOM_INSTRUCTIONS" >> $GITHUB_ENV
echo "EOF" >> $GITHUB_ENV
echo "EXTERNAL_REPOS=$EXTERNAL_REPOS" >> $GITHUB_ENV
echo " ✅ 40 values exported to GITHUB_ENV (28 CodeWiki + 2 YouTube + 3 README + 5 output_paths + 2 custom [with repo context])"
echo ""
# ===================================================================
# 7. SUMMARY
# ===================================================================
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
echo "✅ VALIDATION AND PARSING COMPLETE"
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
echo "📊 Parameters validated: 5"
echo "📦 JSON configurations parsed: 3"
echo "🔢 Individual values extracted: 18"
echo "✅ All values exported and ready for workflow stages"
echo "━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━"
# =========================================================================
# UNIFIED FILE DISCOVERY (Single Source of Truth)
# Discovers SOURCE CODE files and optionally DELETES everything else.
# When SOURCE_FILES_LIMIT > 0:
# 1. Keeps only N source files (.ts, .java, .py, etc.)
# 2. DELETES ALL other files in the repo (aggressive cleanup)
# Generated docs (inline .md) are created AFTER this step, so not affected.
# =========================================================================
- name: Discover Source Files
id: discover_files
env:
SOURCE_FILES_LIMIT: ${{ env.SOURCE_FILES_LIMIT }}
# DOCS_OUTPUT_PATH is needed below so the find can exclude the
# generated docs tree (deleted by the Clean Slate step that runs
# AFTER discovery but BEFORE Stage 1 — source-extension files
# under docs/ would otherwise be enumerated, then deleted, then
# cause ENOENT in Stage 1).
DOCS_OUTPUT_PATH: ${{ env.DOCS_OUTPUT_PATH }}
run: |
source /tmp/workflow-helpers.sh
echo "🔍 Discovering source code files..."
echo " Scanning main repo (.) and dependency repos (../deps/)"
# File paths (all in /tmp to avoid accidental commits)
SOURCE_FILES_LIST="/tmp/.doc-orchestrator-source-files.txt"
ALL_SOURCE_TEMP="/tmp/all_source_files_discovered.txt"
ALL_FILES_TEMP="/tmp/all_files_in_repo.txt"
FILES_TO_DELETE="/tmp/files_to_delete.txt"
# 1. Find SOURCE CODE files only (with standard exclusions)
# IMPORTANT: exclude $DOCS_OUTPUT_PATH/* — the "Clean Slate" step below
# does `rm -rf $DOCS_OUTPUT_PATH` BEFORE Stage 1 reads this list. If
# any source-extension file lives under the docs tree (Doxygen's
# `docs/doxygen/documentation.h`, Sphinx `_extensions/*.py`, etc.),
# it would be enumerated here, then deleted, then Stage 1 hits
# ENOENT trying to read it. This is the source-discovery /
# Clean-Slate / Stage-1 ordering bug — exclusion is the targeted fix.
# The list comes from ci-source.mjs: the served source extensions, the
# served never-source directories, tests skipped by name — the same rule
# the language detection above counted with.
node /tmp/ci-source.mjs list "$ALL_SOURCE_TEMP" --exclude "./$DOCS_OUTPUT_PATH" > /dev/null || { echo "❌ source discovery failed"; exit 1; }
TOTAL_SOURCE=$(wc -l < "$ALL_SOURCE_TEMP" | tr -d ' ')
FILE_LIMIT="${SOURCE_FILES_LIMIT:-0}"
echo " Found $TOTAL_SOURCE source code files"
# Apply limit: keep N source files, DELETE EVERYTHING ELSE
if [ "$FILE_LIMIT" -gt 0 ]; then
echo ""
echo "⚡ DEBUG MODE: Keeping only $FILE_LIMIT source files, deleting EVERYTHING else"
# Keep first N source files
head -n "$FILE_LIMIT" "$ALL_SOURCE_TEMP" > "$SOURCE_FILES_LIST"
KEEPING=$(wc -l < "$SOURCE_FILES_LIST" | tr -d ' ')
# 2. Find ALL files in the repo (except .git and workflow temp files)
find . ../deps 2>/dev/null -type f \
-not -path "*/.git/*" \
-not -path "*/.git" \
-not -name ".doc-orchestrator-*" \
-not -name ".doc-stage*" \
| sort > "$ALL_FILES_TEMP"
TOTAL_FILES=$(wc -l < "$ALL_FILES_TEMP" | tr -d ' ')
echo " Total files in repo: $TOTAL_FILES"
# Build delete list: ALL files EXCEPT the ones we're keeping
# Also preserve workflow temp files (.doc-orchestrator-*, .doc-stage*)
> "$FILES_TO_DELETE"
while IFS= read -r file; do
# Skip workflow temp files we need to preserve
case "$file" in
./.doc-orchestrator-*|./.doc-stage*) continue ;;
esac
# Check if this file is in our keep list
if ! grep -qxF "$file" "$SOURCE_FILES_LIST" 2>/dev/null; then
echo "$file" >> "$FILES_TO_DELETE"
fi
done < "$ALL_FILES_TEMP"
DELETE_COUNT=$(wc -l < "$FILES_TO_DELETE" | tr -d ' ')
echo " Files to delete: $DELETE_COUNT"
# Delete all files NOT in the keep list
DELETED_COUNT=0
while IFS= read -r file_to_delete; do
if [ -f "$file_to_delete" ]; then
rm -f "$file_to_delete"
DELETED_COUNT=$((DELETED_COUNT + 1))
fi
done < "$FILES_TO_DELETE"
echo " ✅ Keeping: $KEEPING source files"
echo " 🗑️ Deleted: $DELETED_COUNT files"
rm -f "$FILES_TO_DELETE" "$ALL_FILES_TEMP"
# Aggressively prune directories (including those with only dotfiles)
echo " 🧹 Pruning directories..."
PRUNED_COUNT=0
# First, delete all dotfiles except in .git and workflow temp files (they prevent dir deletion)
find . -type f -name ".*" \
-not -path "*/.git/*" \
-not -name ".doc-orchestrator-*" \
-not -name ".doc-stage*" \
-delete 2>/dev/null || true
# Multiple passes to handle nested empty directories
for i in 1 2 3 4 5 6 7 8 9 10; do
PASS_COUNT=0
while IFS= read -r empty_dir; do
if [ -d "$empty_dir" ] && [ -z "$(ls -A "$empty_dir" 2>/dev/null)" ]; then
rmdir "$empty_dir" 2>/dev/null && PASS_COUNT=$((PASS_COUNT + 1))
fi
done < <(find . -type d -empty 2>/dev/null | grep -v "^.$" | grep -v ".git")
PRUNED_COUNT=$((PRUNED_COUNT + PASS_COUNT))
[ "$PASS_COUNT" -eq 0 ] && break
done
if [ "$PRUNED_COUNT" -gt 0 ]; then
echo " 🗑️ Pruned $PRUNED_COUNT empty directories"
fi
# Show what's left
echo ""
echo "📂 Remaining structure:"
find . -type d -not -path "*/.git/*" -not -path "*/.git" | head -20
else
# No limit - keep all source files
cp "$ALL_SOURCE_TEMP" "$SOURCE_FILES_LIST"
fi
# Cleanup temp file
rm -f "$ALL_SOURCE_TEMP"
# Count remaining files
FILE_COUNT=$(count_source_files "$SOURCE_FILES_LIST")
MAIN_COUNT=$(count_main_repo_files "$SOURCE_FILES_LIST")
DEPS_COUNT=$(count_dependency_files "$SOURCE_FILES_LIST")
echo ""
echo "📊 Source files to analyze:"
echo " Total: $FILE_COUNT files"
echo " Main repo: $MAIN_COUNT files"
echo " Dependencies: $DEPS_COUNT files"
# Show breakdown by language
echo ""
echo "📋 By language:"
node /tmp/ci-source.mjs breakdown "$SOURCE_FILES_LIST" | sed 's/^/ /'
# Show first 20 files for debugging
echo ""
echo "📄 Sample files (first 20):"
head -20 "$SOURCE_FILES_LIST" | sed 's/^/ /'
# Output for use by subsequent steps
set_output "source_file_count" "$FILE_COUNT"
set_output "source_files_list" "$SOURCE_FILES_LIST"
# =========================================================================
# CREATE PR BRANCH EARLY (Progressive PR Support)
# Creates PR branch and initial PR before any stages run, so each stage
# can commit its results immediately. This prevents data loss on timeouts.
# =========================================================================
- name: Sanitize Branch Name for PR
id: branch-name-early
run: |
source /tmp/workflow-helpers.sh
# Replace colons and other invalid chars with hyphens for git branch name
SAFE_RUN_ID=$(echo "$RUN_ID" | sed 's/[:]/-/g' | sed 's/[^a-zA-Z0-9._-]/-/g')
set_output "safe_run_id" "$SAFE_RUN_ID"
echo "📝 Sanitized RUN_ID for branch: $SAFE_RUN_ID"
- name: Create PR Branch
id: create-pr-branch
run: |
source /tmp/workflow-helpers.sh
echo "📝 Creating PR branch early for progressive commits..."
# Use sanitized run ID for branch name. Branch, PR title, status file and
# commit messages all carry the product name: 🦩 Flamingo Code Documentation.
SAFE_RUN_ID="${{ steps.branch-name-early.outputs.safe_run_id }}"
BRANCH_NAME="docs/flamingo-ai-technical-writer-$SAFE_RUN_ID"
# Configure git
git config user.name "github-actions[bot]"
git config user.email "github-actions[bot]@users.noreply.github.com"
# Create and push empty branch
git checkout -b "$BRANCH_NAME"
# Create initial commit to enable PR creation
echo "# 🦩 Flamingo Code Documentation: Started" > .flamingo-ai-technical-writer-status.md
echo "" >> .flamingo-ai-technical-writer-status.md
echo "Run ID: $SAFE_RUN_ID" >> .flamingo-ai-technical-writer-status.md
echo "Status: In Progress" >> .flamingo-ai-technical-writer-status.md
echo "Started: $(date -u +"%Y-%m-%d %H:%M:%S UTC")" >> .flamingo-ai-technical-writer-status.md
# -f: the status file is a hidden dot-md that many target repos' .gitignore
# patterns (e.g. `.*` / `*status*`) cover — without -f, `git add` fails the
# step. It's removed again in "Cleanup Temporary Files" before the PR.
git add -f .flamingo-ai-technical-writer-status.md
git commit -m "docs: Initialize 🦩 Flamingo Code Documentation run [skip ci]"
git push -u origin "$BRANCH_NAME"
# Store branch name for later steps
set_output "branch_name" "$BRANCH_NAME"
echo "✅ PR branch created: $BRANCH_NAME"
- name: Ensure PR Labels Exist
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: |
echo "🏷️ Ensuring required PR labels exist..."
# Labels used for 🦩 Flamingo Code Documentation PRs
LABELS=(
"documentation:A label for documentation-related PRs:#0075ca"
"automated:PRs created by automation/bots:#ededed"
"in-progress:Work in progress - not ready for merge:#fbca04"
)
for LABEL_DEF in "${LABELS[@]}"; do
LABEL_NAME=$(echo "$LABEL_DEF" | cut -d: -f1)
LABEL_DESC=$(echo "$LABEL_DEF" | cut -d: -f2)
LABEL_COLOR=$(echo "$LABEL_DEF" | cut -d: -f3 | sed 's/#//')
# Check if label exists
if gh label list --json name --jq '.[].name' | grep -q "^${LABEL_NAME}$"; then
echo " ✓ Label '$LABEL_NAME' already exists"
else
echo " + Creating label '$LABEL_NAME'..."
gh label create "$LABEL_NAME" \
--description "$LABEL_DESC" \
--color "$LABEL_COLOR" || true
fi
done
echo "✅ All required labels verified"
- name: Create Initial PR
id: create-initial-pr
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }}
RUN_ID_VAR: ${{ env.RUN_ID }}
REPO_NAME: ${{ github.repository }}
DEFAULT_BRANCH: ${{ github.event.repository.default_branch }}
run: |
source /tmp/workflow-helpers.sh
echo "📝 Creating Pull Request for branch: $BRANCH_NAME"
# Create PR body in a temp file (avoiding YAML parsing issues)
{
echo "🦩 Flamingo Code Documentation: In Progress"
echo ""
echo "Run ID: $RUN_ID_VAR"
echo "Status: Running..."
echo ""
echo "This PR will be updated as each documentation stage completes."
echo ""
echo "Progress"
echo "- Stage 1 Inline Documentation - Starting..."
echo "- Stage 2 Architecture Analysis - Pending"
echo "- Stage 3 Tutorial Generation - Pending"
echo "- Stage 4 Repository Documentation - Pending"
echo ""
echo "Generated by 🦩 Flamingo Code Documentation"
} > /tmp/pr-body.md
# Create PR with gh CLI (works with existing branches)
PR_URL=$(gh pr create \
--base "$DEFAULT_BRANCH" \
--head "$BRANCH_NAME" \
--title "[IN PROGRESS] 🦩 Flamingo Code Documentation" \
--body-file /tmp/pr-body.md \
--label "documentation,automated,in-progress")
# Extract PR number from URL
PR_NUMBER=$(echo "$PR_URL" | grep -oE '[0-9]+$')
echo "✅ PR created: $PR_URL"
echo " PR Number: $PR_NUMBER"
# Set outputs for later steps
set_output "pull-request-url" "$PR_URL"
set_output "pull-request-number" "$PR_NUMBER"
# =========================================================================
# CLEAN SLATE: REMOVE ENTIRE DOCS DIRECTORY
# Deletes ALL documentation before regeneration
# Ensures 100% fresh documentation with zero orphaned files
# All documentation will be regenerated by workflow stages
# =========================================================================
- name: Clean Slate - Remove Regenerated Docs
env:
DOCS_OUTPUT_PATH: ${{ env.DOCS_OUTPUT_PATH }}
REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }}
DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }}
GETTING_STARTED_OUTPUT_PATH: ${{ env.GETTING_STARTED_OUTPUT_PATH }}
DEVELOPMENT_OUTPUT_PATH: ${{ env.DEVELOPMENT_OUTPUT_PATH }}
BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }}
run: |
source /tmp/workflow-helpers.sh
echo "🗑️ Clean Slate: Removing documentation this run will regenerate..."
echo ""
# Recorded so a stage that ends up producing nothing can restore what was
# deleted on its behalf. Without it, a skipped stage turns the pull request
# into a net DELETION of existing documentation.
PRE_CLEAN_SHA=$(git rev-parse HEAD)
echo "PRE_CLEAN_SHA=$PRE_CLEAN_SHA" >> $GITHUB_ENV
echo " Pre-clean commit: $PRE_CLEAN_SHA"
# SCOPED TO THE CONFIGURED STAGES.
#
# This used to `rm -rf $DOCS_OUTPUT_PATH` unconditionally. That was safe only
# while every repo ran all four stages. With per-repo `stages`, wiping the
# whole tree when Stage 2 is disabled means the reference architecture and
# diagrams are deleted and never rebuilt — the pull request becomes a net
# DELETION of existing documentation.
CLEAN_TARGETS=()
if [[ "$STAGES" == *"codewiki"* ]]; then
CLEAN_TARGETS+=("$REFERENCE_OUTPUT_PATH" "$DIAGRAMS_OUTPUT_PATH")
fi
if [[ "$STAGES" == *"tutorials"* ]]; then
CLEAN_TARGETS+=("$GETTING_STARTED_OUTPUT_PATH" "$DEVELOPMENT_OUTPUT_PATH")
fi
# DELETE_TARGETS is what actually gets rm -rf'd; CLEAN_TARGETS is what the
# restore step keys PER-STAGE. They differ only for a full wipe: we delete
# the whole tree (so orphaned files from a previous layout — e.g. synthetic
# module_N dirs left by an earlier clustering-failure run — cannot survive)
# but still RECORD the per-stage subtrees, so the restore leaves each
# subtree deleted iff its OWN stage produced output.
#
# Recording the blanket DOCS_OUTPUT_PATH instead (the old behaviour) made the
# restore treat docs/ as a single unit that is "safe to leave deleted" only
# once ALL FOUR stages complete — but the restore runs right after Stage 2,
# so Stage 3/4 are never 'completed' yet, and it restored the ENTIRE pre-clean
# tree every time, undoing the wipe and resurrecting the orphaned module_N docs.
DELETE_TARGETS=("${CLEAN_TARGETS[@]}")
FULL_WIPE=false
SELECTED_COUNT=$(echo "$STAGES" | tr ',' '\n' | grep -c .)
if [ "$SELECTED_COUNT" -ge "${STAGE_COUNT:-4}" ]; then
FULL_WIPE=true
DELETE_TARGETS=("$DOCS_OUTPUT_PATH")
# Record every stage-owned subtree (NOT the blanket docs/) so the restore
# keys each subtree on its own stage instead of the all-four AND.
CLEAN_TARGETS=("$REFERENCE_OUTPUT_PATH" "$DIAGRAMS_OUTPUT_PATH" "$GETTING_STARTED_OUTPUT_PATH" "$DEVELOPMENT_OUTPUT_PATH")
fi
echo " Stages: $STAGES"
echo " Full wipe: $FULL_WIPE"
echo " Targets: ${CLEAN_TARGETS[*]:-(none)}"
# Recorded so the restore step iterates exactly what was removed — the
# stage -> subtree map has ONE home, here.
{
echo "CLEAN_TARGETS_RECORD<<__EOT__"
for t in ${CLEAN_TARGETS[@]+"${CLEAN_TARGETS[@]}"}; do echo "$t"; done
echo "__EOT__"
} >> $GITHUB_ENV
echo ""
# Count files before deletion (for reporting). Iterate DELETE_TARGETS —
# the actual rm list (blanket docs/ on a full wipe, per-stage subtrees
# otherwise) — not the restore-record CLEAN_TARGETS.
DELETED_FILES=0
for target in ${DELETE_TARGETS[@]+"${DELETE_TARGETS[@]}"}; do
if [ -d "$target" ]; then
TARGET_FILES=$(find "$target" -type f | wc -l | tr -d ' ')
DELETED_FILES=$((DELETED_FILES + TARGET_FILES))
echo " 🗑️ Removing $target ($TARGET_FILES files)..."
rm -rf "$target"
else
echo " ⏭️ $target does not exist (nothing to clean)"
fi
done
if [ ${#DELETE_TARGETS[@]} -eq 0 ]; then
echo " ⏭️ No stage owns a docs subtree — nothing to clean"
fi
# Recreate base directory
mkdir -p "$DOCS_OUTPUT_PATH"
echo " 📁 Recreated empty: $DOCS_OUTPUT_PATH"
# Commit the deletion to git (so it shows in PR)
if [ "$DELETED_FILES" -gt 0 ]; then
echo ""
echo "📝 Committing Clean Slate deletion..."
git add -A
# Check if there are staged changes
STAGED_COUNT=$(git diff --cached --name-only | wc -l | tr -d ' ')
if [ "$STAGED_COUNT" -gt 0 ]; then
git commit -m "chore(docs): Clean slate - remove all documentation ($DELETED_FILES files) [skip ci]"
git push origin "$BRANCH_NAME"
echo " ✅ Committed deletion of $STAGED_COUNT files"
else
echo " ⏭️ No staged changes (docs already clean)"
fi
fi
echo ""
echo "✅ Clean slate complete - ready for fresh documentation generation"
echo ""
# =========================================================================
# STAGE 1: INLINE CLASS DOCUMENTATION
# Generate .md files next to each source class
# Uses discovered files from the unified file discovery step
# =========================================================================
- name: Setup Node.js
uses: actions/setup-node@v4
with:
node-version: '22'
# =========================================================================
# STAGE 0: CODE GRAPH (same build as the standalone code-graph job above)
# Tags the SOURCE branch head so the hub can render this run's
# ecosystem.md from the snapshot of the commit being documented. The hub
# promotes to `live` only when the source branch is the default branch.
# Never fatal: a graph failure costs cross-repo facts, not the docs run.
# =========================================================================
# The command is CODE_GRAPH_INSTALL_COMMAND (lib/config/code-graph-workflow.ts).
- name: Install graph dependencies
continue-on-error: true
run: mkdir -p "$RUNNER_TEMP/code-graph-deps" && cd "$RUNNER_TEMP/code-graph-deps" && printf '%s' '{"name":"code-graph-deps","version":"1.0.0","private":true,"dependencies":{"web-tree-sitter":"0.27.0","@vscode/tree-sitter-wasm":"0.3.1","yaml":"2.9.1"}}' > package.json && printf '%s' '{"name":"code-graph-deps","version":"1.0.0","lockfileVersion":3,"requires":true,"packages":{"":{"name":"code-graph-deps","version":"1.0.0","dependencies":{"web-tree-sitter":"0.27.0","@vscode/tree-sitter-wasm":"0.3.1","yaml":"2.9.1"}},"node_modules/web-tree-sitter":{"version":"0.27.0","resolved":"https://registry.npmjs.org/web-tree-sitter/-/web-tree-sitter-0.27.0.tgz","integrity":"sha512-XK08gj6RwTMQatAG7uVRP8MunqotL/XC19vHgkSPKmELgbGPBj4ECvB8haHOUnyj6ls2B8t42UTro14zxGgAHg=="},"node_modules/@vscode/tree-sitter-wasm":{"version":"0.3.1","resolved":"https://registry.npmjs.org/@vscode/tree-sitter-wasm/-/tree-sitter-wasm-0.3.1.tgz","integrity":"sha512-RJFoomET6FajjG511fmQxeBQfU6M24a0aFZPqpid+ttIxanWf1VGytBG0UmsGjt07qmIPJS8U31D+aecuCucsQ=="},"node_modules/yaml":{"version":"2.9.1","resolved":"https://registry.npmjs.org/yaml/-/yaml-2.9.1.tgz","integrity":"sha512-3NxN8+78OdzbT7C/WjGsyfPAtJaN3FNDsWxv7Y7mcDsT/oOmgW8BpyQQFFBnvZE3j9Y2Sdz1ULFLezL7Eb2yFw=="}}}' > package-lock.json && npm ci --ignore-scripts --no-audit --no-fund && echo "CODE_GRAPH_DEPS_DIR=$RUNNER_TEMP/code-graph-deps" >> "$GITHUB_ENV" || { echo "::warning::graph dependencies failed their lockfile-enforced install; continuing without them"; rm -rf "$RUNNER_TEMP/code-graph-deps"; exit 1; }
- name: Build and upload the code graph
id: graph
continue-on-error: true
env:
WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }}
CODE_GRAPH_DEPS_DIR: ${{ env.CODE_GRAPH_DEPS_DIR }}
GITHUB_REPOSITORY: ${{ github.repository }}
CODE_GRAPH_BRANCH: ${{ env.SOURCE_BRANCH }}
CODE_GRAPH_COMMIT_SHA: ${{ env.SOURCE_HEAD_SHA }}
run: node /tmp/code-graph-build.mjs
- name: Install Stage 1 Dependencies
if: contains(env.STAGES, 'inline-docs')
# Install generator deps in an ISOLATED tree under RUNNER_TEMP, NOT the target
# repo. npm resolves against an empty package.json here, so a target repo's own
# peer conflicts (e.g. react-accessible-accordion vs react 18) can never make this
# fail. No --legacy-peer-deps / --no-save band-aids. Generators find these via the
# NODE_PATH set on the generate step (RUNNER_TEMP/doc-orch-deps/node_modules).
run: |
mkdir -p "$RUNNER_TEMP/doc-orch-deps" && cd "$RUNNER_TEMP/doc-orch-deps"
npm init -y >/dev/null 2>&1
npm install @anthropic-ai/sdk@0.115.0 zod@3.25.76 glob@13.0.6
- name: Generate Inline Class Documentation
id: stage1
if: contains(env.STAGES, 'inline-docs')
env:
# SECURITY: Pass secrets per-step with inline masking.
# No ANTHROPIC_API_KEY — this stage calls Claude through the hub
# (/api/ci/claude), which the WEBHOOK_SECRET below authenticates.
WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }}
# Claude model SSOT — see workflow env CLAUDE_MODEL block
CLAUDE_MODEL: ${{ env.CLAUDE_MODEL }}
DOCS_OUTPUT_PATH: ${{ env.DOCS_OUTPUT_PATH }}
# Stage timeout (in hours)
STAGE1_TIMEOUT_HOURS: ${{ env.STAGE1_TIMEOUT_HOURS }}
# Incremental commit + push every N generated docs (see workflow env)
STAGE1_PUSH_INTERVAL: ${{ env.STAGE1_PUSH_INTERVAL }}
BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }}
STAGE1_PUSH_MARKER: /tmp/stage1-progress-pushes
# Unified file discovery result (single source of truth)
SOURCE_FILES_LIST: ${{ steps.discover_files.outputs.source_files_list }}
SOURCE_FILE_COUNT: ${{ steps.discover_files.outputs.source_file_count }}
# NODE_PATH to find modules from /tmp/ scripts
NODE_PATH: ${{ runner.temp }}/doc-orch-deps/node_modules
# Custom AI Instructions (All Stages)
CUSTOM_REPO_INSTRUCTIONS: ${{ env.CUSTOM_INSTRUCTIONS }}
# External Repositories
EXTERNAL_REPOS: ${{ env.EXTERNAL_REPOS }}
run: |
source /tmp/workflow-helpers.sh
echo "📝 Stage 1: Generating inline documentation..."
echo " Using unified file list: $SOURCE_FILES_LIST ($SOURCE_FILE_COUNT files)"
echo " Incremental push: every $STAGE1_PUSH_INTERVAL generated docs → $BRANCH_NAME"
# Fresh marker: the generator appends one line per progress push, and the
# commit step below reads it to know work was already pushed.
rm -f "$STAGE1_PUSH_MARKER"
# Script already downloaded to /tmp/ in setup step. run_stage records
# the outcome as stage1_status — see its note in workflow-helpers.sh.
run_stage "Stage 1" "$STAGE1_TIMEOUT_HOURS" stage1_status node /tmp/generate-inline-docs.cjs
# Count generated files (hidden .*.md files)
INLINE_DOCS=$(find . -name ".*.md" -newer .git -type f -not -path "./node_modules/*" -not -path "./.git/*" | wc -l)
set_output "stage1_files" "$INLINE_DOCS"
# =========================================================================
# COMMIT STAGE 1 RESULTS (Progressive PR)
# =========================================================================
# `!= ''`, not `== 'completed'`: runs on a FAILED stage too — see run_stage
# in workflow-helpers.sh (partial output is worth committing; the status
# is what reports the truth home). The build gate holds every run_stage
# commit step to this predicate.
- name: Commit and Push Stage 1 Results
if: always() && steps.stage1.outputs.stage1_status != ''
id: commit-stage1
env:
BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }}
STAGE_FILES: ${{ steps.stage1.outputs.stage1_files }}
STAGE1_PUSH_MARKER: /tmp/stage1-progress-pushes
run: |
source /tmp/workflow-helpers.sh
echo "📝 Committing Stage 1 (Inline Documentation) results..."
# The generator already commits + pushes every N docs (STAGE1_PUSH_INTERVAL).
# Report how much landed that way; what's left here is the final partial batch.
if [ -s "$STAGE1_PUSH_MARKER" ]; then
PUSHED_BATCHES=$(wc -l < "$STAGE1_PUSH_MARKER" | tr -d ' ')
PUSHED_FILES=$(awk '{ sum += $1 } END { print sum + 0 }' "$STAGE1_PUSH_MARKER")
echo " Already pushed during generation: $PUSHED_FILES files in $PUSHED_BATCHES batches"
fi
# Push any commits the generator made but could not push (transient push failure)
git push origin "HEAD:refs/heads/$BRANCH_NAME" 2>/dev/null || true
# Stage all .md files generated by Stage 1 (hidden inline docs)
find . -name ".*.md" -type f \
-not -path "./node_modules/*" \
-not -path "./.git/*" \
-exec git add -f {} \; 2>/dev/null || true
# Check if there are changes
STAGED_COUNT=$(git diff --cached --name-only | wc -l)
if [ "$STAGED_COUNT" -gt 0 ]; then
# Commit and push
git commit -m "docs: Stage 1 - Inline documentation ($STAGE_FILES files) [skip ci]"
git push origin "$BRANCH_NAME"
echo "✅ Committed and pushed $STAGED_COUNT files from Stage 1"
set_output "committed" "true"
elif [ -s "$STAGE1_PUSH_MARKER" ]; then
# Everything already landed via the incremental progress pushes
echo "✅ Nothing left to commit: all Stage 1 files pushed during generation"
set_output "committed" "true"
else
echo "⚠️ No Stage 1 files to commit"
set_output "committed" "false"
fi
# =========================================================================
# ORPHAN DETECTION: INLINE FILES ONLY
# Runs immediately after Stage 1 to clean up orphaned inline docs
# Orphan = .*.md file where source file was deleted
# =========================================================================
# `== 'completed'` HERE IS DELIBERATE, unlike the commit steps: deleting
# "orphaned" docs after a stage that FAILED would delete docs whose
# sources were never re-examined.
- name: Clean Up Orphaned Inline Documentation Files
if: always() && steps.stage1.outputs.stage1_status == 'completed'
id: orphan-detection-inline
env:
BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }}
run: |
source /tmp/workflow-helpers.sh
# Run orphan detection script
bash /tmp/detect-orphans.sh || true
# Check if orphans were deleted (script writes list to /tmp/orphaned-inline-files.txt)
DELETED_FILES_LIST="/tmp/orphaned-inline-files.txt"
if [ -f "$DELETED_FILES_LIST" ] && [ -s "$DELETED_FILES_LIST" ]; then
DELETED_COUNT=$(wc -l < "$DELETED_FILES_LIST" | tr -d ' ')
echo ""
echo "📝 Committing $DELETED_COUNT orphan deletions..."
# Only add the specific files that were deleted by the script
while IFS= read -r deleted_file; do
git add "$deleted_file" 2>/dev/null || true
done < "$DELETED_FILES_LIST"
# Verify we have staged changes
STAGED_COUNT=$(git diff --cached --name-only | wc -l | tr -d ' ')
if [ "$STAGED_COUNT" -gt 0 ]; then
git commit -m "chore(docs): Remove $DELETED_COUNT orphaned inline files [skip ci]"
git push origin "$BRANCH_NAME"
echo "✅ Committed $STAGED_COUNT orphan deletions"
set_output "orphans_deleted" "$STAGED_COUNT"
else
echo "⚠️ No changes to commit (files may have been already removed)"
set_output "orphans_deleted" "0"
fi
else
echo "✅ No orphaned inline files detected"
set_output "orphans_deleted" "0"
fi
- name: Update PR with Stage 1 Progress
if: always() && steps.commit-stage1.outputs.committed == 'true' && steps.create-initial-pr.outputs.pull-request-number
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }}
STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files }}
RUN_ID_VAR: ${{ env.RUN_ID }}
REPO_NAME: ${{ github.repository }}
run: |
source /tmp/workflow-helpers.sh
echo "📝 Updating PR description with Stage 1 results..."
# Update PR body
PR_BODY="## 🦩 Flamingo Code Documentation: In Progress
**Run ID:** \`$RUN_ID_VAR\`
**Status:** 🔄 Running...
This PR is being updated as each documentation stage completes.
### Progress
- ✅ Stage 1: Inline Documentation - Completed ($STAGE1_FILES files)
- ⏳ Stage 2: Architecture Analysis - Running...
- ⏱️ Stage 3: Tutorial Generation - Pending
- ⏱️ Stage 4: Repository Documentation - Pending
---
🦩 Generated by [Flamingo Code Documentation](https://flamingo.run)"
gh pr edit "$PR_NUMBER" --body "$PR_BODY"
echo "✅ PR #$PR_NUMBER updated with Stage 1 progress"
- name: Report Stage 1 Progress
if: always() && contains(env.STAGES, 'inline-docs') && env.HUB_BASE_URL != ''
continue-on-error: true
env:
WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }}
WORKFLOW_RUN_ID: ${{ github.run_id }}
WORKFLOW_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }}
STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }}
PR_URL: ${{ steps.create-initial-pr.outputs.pull-request-url }}
PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }}
run: |
source /tmp/workflow-helpers.sh
# Construct callback URL from hub base URL
CALLBACK_URL="${HUB_BASE_URL}/api/doc-orchestrator/webhook"
echo "📤 Reporting Stage 1 (Inline Docs) completion with PR URL..."
report_stage_progress "$CALLBACK_URL" "$WEBHOOK_SECRET" "$RUN_ID" "$REPO_ID" \
"$WORKFLOW_RUN_ID" "$WORKFLOW_URL" "codewiki" \
"$STAGE1_STATUS" "$STAGE1_FILES" "" "0" "" "0" "" "0" \
"$PR_URL" "$PR_NUMBER"
# =========================================================================
# ECOSYSTEM FACTS — derived by the hub from the code graph, never written
# by a model. Two renderings of the same live snapshot: ecosystem.md
# (committed under the reference tree and fed to the Stage 2/3/4 prompts
# as ground truth for the Dependencies sections) and the marker-delimited
# AGENTS.md block (upserted in place, idempotent). A repo with no graph
# yet is a notice, not a failure.
# =========================================================================
- name: Fetch ecosystem facts
continue-on-error: true
env:
WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }}
REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }}
run: |
source /tmp/workflow-helpers.sh
# Through ci-hub.mjs, the shell's way into the ONE hub transport
# (code-review-lib.mjs): it checks the destination before the secret
# leaves, keeps the secret in a header, and writes the body only on a
# 2xx. It prints the status and exits 0 whenever the hub ANSWERED — a 404
# is an answer ("no graph yet"), not a failure.
REPO_PARAM=$(printf '%s' "$GITHUB_REPOSITORY" | sed 's|/|%2F|g')
ECOSYSTEM_PATH="/api/ci/code-graph/ecosystem.md?repo=${REPO_PARAM}"
HTTP_CODE=$(node /tmp/ci-hub.mjs get "$ECOSYSTEM_PATH" /tmp/ecosystem.md) || HTTP_CODE="000"
if [ "$HTTP_CODE" = "404" ]; then
echo "::notice::no code graph yet for $GITHUB_REPOSITORY — skipping ecosystem facts"
rm -f /tmp/ecosystem.md
exit 0
fi
if [ "$HTTP_CODE" != "200" ]; then
echo "::warning::ecosystem facts request failed (HTTP $HTTP_CODE) — the Dependencies sections run without cross-repo facts"
rm -f /tmp/ecosystem.md
exit 0
fi
# Kept in /tmp ONLY until Stage 2 has run. The reference directory is
# CodeWiki's output directory, and CodeWiki asks "already contains
# documentation. Overwrite?" when it finds a .md file there — a prompt
# a runner cannot answer, so Stage 2 aborted on every CodeWiki repo
# (CodeWiki run 35293807658). The Stage 2 commit step copies the file
# into place, after either engine has written its own output.
echo "✅ ecosystem.md ($(wc -c < /tmp/ecosystem.md | tr -d ' ') bytes) fetched; copied to $REFERENCE_OUTPUT_PATH after Stage 2"
HTTP_CODE=$(node /tmp/ci-hub.mjs get "${ECOSYSTEM_PATH}&format=agents" /tmp/ecosystem-agents.md) || HTTP_CODE="000"
if [ "$HTTP_CODE" = "200" ]; then
upsert_marker_block AGENTS.md /tmp/ecosystem-agents.md
else
echo "::warning::AGENTS.md block request failed (HTTP $HTTP_CODE) — AGENTS.md left untouched"
rm -f /tmp/ecosystem-agents.md
fi
# =========================================================================
# STAGE 2: CODEWIKI ANALYSIS
# Generate architecture overview and module tree
# Language detection already ran before Stage 1 (uses steps.detect_language outputs)
# =========================================================================
- name: Setup Python 3.12
if: contains(env.STAGES, 'codewiki') && steps.detect_language.outputs.codewiki_supported == 'true'
uses: actions/setup-python@v5
with:
python-version: '3.12'
- name: Install CodeWiki
id: codewiki_install
if: contains(env.STAGES, 'codewiki') && steps.detect_language.outputs.codewiki_supported == 'true'
continue-on-error: true
env:
CODEWIKI_MAX_FILES_PER_MODULE: ${{ env.CODEWIKI_MAX_FILES_PER_MODULE }}
CODEWIKI_REPO: ${{ env.CODEWIKI_REPO }}
run: |
# Install keyrings.alt for headless keyring support in CI environments
# Install ipython to suppress "Mermaidjs magic function not available" warning
# Install colorama for CodeWiki colored terminal output
pip install keyrings.alt ipython colorama
# Clone CodeWiki directly (no pip caching issues)
# Fixes baked into fork:
# - retries=3 for Pydantic AI agents (prevents "Tool exceeded max retries count of 1")
# - Synthetic module creation when clustering returns 0 modules (prevents context overflow)
# - 'children' key fix for synthetic modules
# - module_tree.json path fix (commit c1dfe5c) - loads from base docs dir, not nested module dir
# See: https://github.com/flamingo-stack/CodeWiki
# Extract repo URL from CODEWIKI_REPO (strip git+ prefix and @branch/commit suffix)
REPO_URL=$(echo "$CODEWIKI_REPO" | sed 's|^git+||' | sed 's|@[^@]*$||')
REF=$(echo "$CODEWIKI_REPO" | grep -o '@[^@]*$' | sed 's|^@||' || echo "main")
echo "📦 Cloning CodeWiki from: $REPO_URL (ref: ${REF:-main})"
rm -rf /tmp/CodeWiki
# Clone and checkout - handle both branches and commit hashes
if [[ "${REF}" =~ ^[0-9a-f]{7,40}$ ]]; then
# Commit hash - clone full repo and checkout specific commit
git clone "$REPO_URL" /tmp/CodeWiki
cd /tmp/CodeWiki && git checkout "${REF}" && cd -
else
# Branch name - shallow clone
git clone --depth 1 --branch "${REF:-main}" "$REPO_URL" /tmp/CodeWiki
fi
echo " Commit: $(cd /tmp/CodeWiki && git rev-parse --short HEAD)"
# Install from local clone (reliable, no caching)
echo "📦 Installing CodeWiki from local clone..."
pip install --no-cache-dir /tmp/CodeWiki
source /tmp/workflow-helpers.sh
set_output "codewiki_installed" "true"
- name: Configure CodeWiki
id: codewiki_config
if: contains(env.STAGES, 'codewiki') && steps.detect_language.outputs.codewiki_supported == 'true' && steps.codewiki_install.outputs.codewiki_installed == 'true'
continue-on-error: true
env:
# SECURITY: Pass secrets per-step with inline masking
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
# Set keyring backend via env var (must be set before any keyring operations)
PYTHON_KEYRING_BACKEND: keyrings.alt.file.PlaintextKeyring
# Flamingo Markdown Guidelines path (needed for module import during config/validate)
FLAMINGO_MARKDOWN_GUIDELINES_PATH: /tmp/flamingo-markdown-guidelines.md
# OSS Tenant Structure: Stage 2 outputs (for clean slate deletion)
REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }}
DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }}
run: |
source /tmp/workflow-helpers.sh
# === KEYRING CONFIGURATION FOR CI ===
# CodeWiki stores API keys in system keyring. In CI (no GUI), we must:
# 1. Create keyring config to specify PlaintextKeyring backend
# 2. Create data directory for credential storage
# See: https://github.com/FSoft-AI4Code/CodeWiki - uses keyring.set_password()
echo "🔑 Setting up keyring for headless CI environment..."
# Create keyring configuration directory and config file
mkdir -p ~/.config/python_keyring
cat > ~/.config/python_keyring/keyringrc.cfg << 'KEYRING_CFG'
[backend]
default-keyring=keyrings.alt.file.PlaintextKeyring
KEYRING_CFG
# Ensure keyring data directory exists with proper permissions
mkdir -p ~/.local/share/python_keyring
chmod 700 ~/.local/share/python_keyring
# Debug: Verify keyring is properly configured
echo "📋 Keyring backend verification:"
python3 -c "import keyring; print(f' Active backend: {keyring.get_keyring()}')"
# Configure CodeWiki with separate cluster and generation providers/models
# CodeWiki calls provider APIs directly via --base-url
# Model names should match the provider's API format (no LiteLLM prefix needed)
# OpenAI: gpt-4o, gpt-4-turbo, gpt-4o-mini
# Provider ids come from MODEL_METADATA in lib/constants/ai-models.ts
# See: https://github.com/FSoft-AI4Code/CodeWiki
echo "🔧 Configuring CodeWiki..."
echo " Cluster (Phase 2): $CODEWIKI_CLUSTER_PROVIDER / $CODEWIKI_CLUSTER_MODEL"
echo " Generation (Phase 3+): $CODEWIKI_GENERATION_PROVIDER / $CODEWIKI_GENERATION_MODEL"
# Source helper functions for configuration
source /tmp/workflow-helpers.sh
# Determine API keys for each provider (cluster, generation/main, fallback)
# Each provider can use a different AI service (OpenAI, Anthropic, etc.)
if [ "$CODEWIKI_CLUSTER_PROVIDER" = "anthropic" ]; then
CLUSTER_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}"
echo " Cluster: Using Anthropic API key"
else
CLUSTER_API_KEY="${{ secrets.OPENAI_API_KEY }}"
echo " Cluster: Using OpenAI API key"
fi
if [ "$CODEWIKI_GENERATION_PROVIDER" = "anthropic" ]; then
MAIN_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}"
echo " Generation: Using Anthropic API key"
else
MAIN_API_KEY="${{ secrets.OPENAI_API_KEY }}"
echo " Generation: Using OpenAI API key"
fi
if [ "$CODEWIKI_FALLBACK_PROVIDER" = "anthropic" ]; then
FALLBACK_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}"
echo " Fallback: Using Anthropic API key"
else
FALLBACK_API_KEY="${{ secrets.OPENAI_API_KEY }}"
echo " Fallback: Using OpenAI API key"
fi
# Configure CodeWiki from CODEWIKI_CONFIG_JSON (single extractor: configure_codewiki_from_json)
# Pass per-provider API keys for mixed provider configurations
configure_codewiki_from_json "$CODEWIKI_CONFIG_JSON" "$CLUSTER_API_KEY" "$MAIN_API_KEY" "$FALLBACK_API_KEY"
if [ $? -ne 0 ]; then
echo "❌ CodeWiki configuration failed"
exit 1
fi
# Set environment variables for backward compatibility with run-codewiki-analysis.sh
export MAIN_MODEL="$CODEWIKI_GENERATION_MODEL"
export FALLBACK_MODEL_1="$CODEWIKI_FALLBACK_MODEL"
if [ "$PRIMARY_PROVIDER" = "anthropic" ]; then
export ANTHROPIC_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}"
else
export OPENAI_API_KEY="${{ secrets.OPENAI_API_KEY }}"
fi
# Verify configuration was saved
echo ""
echo "📋 CodeWiki configuration:"
python -m codewiki config show
echo ""
echo "✅ Validating configuration..."
python -m codewiki config validate
echo ""
set_output "codewiki_configured" "true"
- name: Run CodeWiki Analysis
id: stage2
if: contains(env.STAGES, 'codewiki') && steps.detect_language.outputs.codewiki_supported == 'true' && steps.codewiki_config.outputs.codewiki_configured == 'true'
continue-on-error: false
env:
# Keyring backend for CI (must match config step)
PYTHON_KEYRING_BACKEND: keyrings.alt.file.PlaintextKeyring
# API keys for both providers (CodeWiki will use the one configured)
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
# OSS Tenant Structure: Stage 2 outputs
REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }}
DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }}
# Stage timeout
STAGE2_TIMEOUT_HOURS: ${{ env.STAGE2_TIMEOUT_HOURS }}
# CodeWiki JSON configuration (required for unified function)
CODEWIKI_CONFIG_JSON: ${{ env.CODEWIKI_CONFIG_JSON }}
# CodeWiki model configuration (cluster, generation, fallback)
CODEWIKI_CLUSTER_PROVIDER: ${{ env.CODEWIKI_CLUSTER_PROVIDER }}
CODEWIKI_CLUSTER_MODEL: ${{ env.CODEWIKI_CLUSTER_MODEL }}
CODEWIKI_CLUSTER_MAX_TOKENS: ${{ env.CODEWIKI_CLUSTER_MAX_TOKENS }}
CODEWIKI_CLUSTER_MAX_TOKEN_FIELD: ${{ env.CODEWIKI_CLUSTER_MAX_TOKEN_FIELD }}
CODEWIKI_GENERATION_PROVIDER: ${{ env.CODEWIKI_GENERATION_PROVIDER }}
CODEWIKI_GENERATION_MODEL: ${{ env.CODEWIKI_GENERATION_MODEL }}
CODEWIKI_GENERATION_MAX_TOKENS: ${{ env.CODEWIKI_GENERATION_MAX_TOKENS }}
CODEWIKI_GENERATION_MAX_TOKEN_FIELD: ${{ env.CODEWIKI_GENERATION_MAX_TOKEN_FIELD }}
CODEWIKI_FALLBACK_PROVIDER: ${{ env.CODEWIKI_FALLBACK_PROVIDER }}
CODEWIKI_FALLBACK_MODEL: ${{ env.CODEWIKI_FALLBACK_MODEL }}
CODEWIKI_FALLBACK_MAX_TOKENS: ${{ env.CODEWIKI_FALLBACK_MAX_TOKENS }}
CODEWIKI_FALLBACK_MAX_TOKEN_FIELD: ${{ env.CODEWIKI_FALLBACK_MAX_TOKEN_FIELD }}
CODEWIKI_MAX_DEPTH: ${{ env.CODEWIKI_MAX_DEPTH }}
# Flamingo Markdown Guidelines path for CodeWiki prompts
FLAMINGO_MARKDOWN_GUIDELINES_PATH: /tmp/flamingo-markdown-guidelines.md
# Markdown Validation Rules (injected into all prompts)
VALIDATION_RULES_PATH: ${{ env.VALIDATION_RULES_PATH }}
# Custom AI Instructions (All Stages)
CUSTOM_REPO_INSTRUCTIONS: ${{ env.CUSTOM_INSTRUCTIONS }}
# External Repositories
EXTERNAL_REPOS: ${{ env.EXTERNAL_REPOS }}
# Dependencies (for CodeWiki multi-path support)
DEPENDENCIES: ${{ env.DEPENDENCIES }}
run: |
# Determine per-provider API keys (same logic as Configure step)
if [ "$CODEWIKI_CLUSTER_PROVIDER" = "anthropic" ]; then
export CLUSTER_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}"
else
export CLUSTER_API_KEY="${{ secrets.OPENAI_API_KEY }}"
fi
if [ "$CODEWIKI_GENERATION_PROVIDER" = "anthropic" ]; then
export MAIN_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}"
else
export MAIN_API_KEY="${{ secrets.OPENAI_API_KEY }}"
fi
if [ "$CODEWIKI_FALLBACK_PROVIDER" = "anthropic" ]; then
export FALLBACK_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}"
else
export FALLBACK_API_KEY="${{ secrets.OPENAI_API_KEY }}"
fi
# Verify dependencies directory before CodeWiki runs
echo ""
echo "🔍 Pre-CodeWiki Dependency Verification:"
echo " Current directory: $(pwd)"
echo " Absolute path: $(realpath .)"
echo ""
if [ -d "./deps" ]; then
echo " ✅ ./deps EXISTS"
echo " Contents: $(ls -1 ./deps 2>/dev/null | wc -l) repositories"
ls -la ./deps 2>/dev/null | head -5
else
echo " ❌ ./deps NOT FOUND"
fi
if [ -d "../deps" ]; then
echo " ✅ ../deps EXISTS"
echo " Absolute: $(realpath ../deps)"
echo " Contents: $(ls -1 ../deps 2>/dev/null | wc -l) repositories"
ls -la ../deps 2>/dev/null | head -5
else
echo " ❌ ../deps NOT FOUND"
fi
echo " DEPENDENCIES env: ${DEPENDENCIES:-<empty>}"
echo ""
# Run externalized CodeWiki analysis script
/tmp/run-codewiki-analysis.sh
# Alternative: Claude Architecture Analysis for ALL languages (when CodeWiki is not supported)
- name: Run Claude Architecture Analysis (All Languages)
id: stage2_alt
if: contains(env.STAGES, 'codewiki') && steps.detect_language.outputs.codewiki_supported == 'false'
env:
# No ANTHROPIC_API_KEY — this stage calls Claude through the hub
# (/api/ci/claude), which the WEBHOOK_SECRET below authenticates.
WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }}
# SSOT — see workflow env CLAUDE_MODEL block
CLAUDE_MODEL: ${{ env.CLAUDE_MODEL }}
PRIMARY_LANGUAGE: ${{ steps.detect_language.outputs.primary_language }}
# OSS Tenant Structure: Stage 2 outputs
REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }}
DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }}
# Unified file discovery result (single source of truth)
SOURCE_FILES_LIST: ${{ steps.discover_files.outputs.source_files_list }}
SOURCE_FILE_COUNT: ${{ steps.discover_files.outputs.source_file_count }}
# Custom AI Instructions (All Stages)
CUSTOM_REPO_INSTRUCTIONS: ${{ env.CUSTOM_INSTRUCTIONS }}
# External Repositories
EXTERNAL_REPOS: ${{ env.EXTERNAL_REPOS }}
# Dependencies (for consistency with other stages)
DEPENDENCIES: ${{ env.DEPENDENCIES }}
run: |
# Run externalized Claude architecture analysis script
/tmp/run-claude-architecture-analysis.sh
# =========================================================================
# RESTORE STAGE 2 DOCS WHEN STAGE 2 PRODUCED NOTHING
# Clean Slate removed the reference/diagrams trees because `codewiki` was in
# STAGES. If neither Stage-2 variant then completed — a repo with no
# discoverable source, an install failure, a skip — the deletion would be the
# only Stage-2 change in the pull request, i.e. a net removal of documentation
# nobody asked to remove. Put it back.
# =========================================================================
- name: Restore Docs That No Stage Regenerated
if: always()
env:
BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }}
STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status }}
STAGE2_STATUS: ${{ steps.stage2.outputs.stage2_status }}
STAGE2_ALT_STATUS: ${{ steps.stage2_alt.outputs.stage2_status }}
run: |
# Clean Slate deleted whatever the configured stages own, and committed that
# deletion. Any owned subtree whose stage then produced nothing must be put
# back — otherwise the pull request is a net REMOVAL of documentation nobody
# asked to remove. Iterates the exact list Clean Slate recorded, so the
# stage -> subtree map has one home.
if [ -z "$CLEAN_TARGETS_RECORD" ]; then
echo "⏭️ Clean Slate removed nothing — no restore needed"
exit 0
fi
STAGE2_OK=false
[ "$STAGE2_STATUS" = "completed" ] && STAGE2_OK=true
[ "$STAGE2_ALT_STATUS" = "completed" ] && STAGE2_OK=true
# CodeWiki leaves multi-GB scratch behind on a failed run; never let it near
# the index.
rm -rf "$REFERENCE_OUTPUT_PATH/temp" "$DIAGRAMS_OUTPUT_PATH/temp" 2>/dev/null || true
RESTORED=0
while IFS= read -r target; do
[ -n "$target" ] || continue
# A target is safe to leave deleted only if something regenerated it.
case "$target" in
"$REFERENCE_OUTPUT_PATH"|"$DIAGRAMS_OUTPUT_PATH")
[ "$STAGE2_OK" = true ] && continue ;;
"$GETTING_STARTED_OUTPUT_PATH"|"$DEVELOPMENT_OUTPUT_PATH")
[ "$STAGE3_STATUS" = "completed" ] && continue ;;
"$DOCS_OUTPUT_PATH")
# Full wipe: only fully safe when every stage delivered.
if [ "$STAGE1_STATUS" = "completed" ] && [ "$STAGE2_OK" = true ] && \
[ "$STAGE3_STATUS" = "completed" ] && [ "$STAGE4_STATUS" = "completed" ]; then
continue
fi ;;
esac
if git checkout "$PRE_CLEAN_SHA" -- "$target" 2>/dev/null; then
echo " ↩️ Restored $target (no stage regenerated it)"
RESTORED=1
fi
done <<< "$CLEAN_TARGETS_RECORD"
# `git checkout -- <path>` already stages exactly those paths. Deliberately NO
# `git add -A`: at this point the workspace holds npm install output from
# Stage 1 and, on a failed CodeWiki run, its scratch trees.
if [ "$RESTORED" -eq 1 ] && [ -n "$(git diff --cached --name-only)" ]; then
git commit -m "chore(docs): restore documentation no stage regenerated [skip ci]"
git push origin "$BRANCH_NAME"
echo " ✅ Restore committed"
else
echo " ⏭️ Nothing to restore"
fi
# =========================================================================
# COMMIT STAGE 2 RESULTS (Progressive PR)
# =========================================================================
- name: Commit and Push Stage 2 Results
if: always() && (steps.stage2.outputs.stage2_status == 'completed' || steps.stage2_alt.outputs.stage2_status == 'completed')
id: commit-stage2
env:
BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }}
STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files }}
REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }}
DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }}
run: |
source /tmp/workflow-helpers.sh
echo "📝 Committing Stage 2 (Architecture Analysis) results..."
# Clean up CodeWiki temp files from all output directories
echo "🧹 Cleaning up temporary files..."
rm -rf "$REFERENCE_OUTPUT_PATH/temp" 2>/dev/null || echo "⚠️ Warning: Could not remove $REFERENCE_OUTPUT_PATH/temp"
rm -rf "$DIAGRAMS_OUTPUT_PATH/temp" 2>/dev/null || echo "⚠️ Warning: Could not remove $DIAGRAMS_OUTPUT_PATH/temp"
# Verify cleanup
if [ -d "$REFERENCE_OUTPUT_PATH/temp" ]; then
echo "❌ ERROR: Failed to cleanup temp files in $REFERENCE_OUTPUT_PATH/temp"
ls -la "$REFERENCE_OUTPUT_PATH/temp"
exit 1
fi
# Create .gitignore in CodeWiki output directories to prevent temp files from being committed
echo "📝 Creating .gitignore files in output directories..."
for output_dir in "$REFERENCE_OUTPUT_PATH" "$DIAGRAMS_OUTPUT_PATH"; do
if [ -d "$output_dir" ]; then
{
echo "# CodeWiki temp files (dependency graphs can be 7GB+)"
echo "temp/"
echo "dependency_graphs/"
echo ""
echo "# JSON intermediate files (except schema/config)"
echo "*.json"
echo "!*-schema.json"
echo "!*-config.json"
} > "$output_dir/.gitignore"
echo " ✅ Created $output_dir/.gitignore"
fi
done
# Debug: Show what files exist before staging
echo ""
echo "🔍 Debug: Files in Stage 2 output directories BEFORE staging:"
echo " Reference directory ($REFERENCE_OUTPUT_PATH):"
if [ -d "$REFERENCE_OUTPUT_PATH" ]; then
find "$REFERENCE_OUTPUT_PATH" -type f \( -name "*.md" -o -name "*.mmd" -o -name ".gitignore" \) | head -20
FILE_COUNT=$(find "$REFERENCE_OUTPUT_PATH" -type f \( -name "*.md" -o -name "*.mmd" \) | wc -l | tr -d ' ')
echo " Total .md/.mmd files: $FILE_COUNT"
else
echo " ⚠️ Directory does not exist!"
fi
echo ""
echo " Diagrams directory ($DIAGRAMS_OUTPUT_PATH):"
if [ -d "$DIAGRAMS_OUTPUT_PATH" ]; then
find "$DIAGRAMS_OUTPUT_PATH" -type f \( -name "*.md" -o -name "*.mmd" -o -name ".gitignore" \) | head -20
FILE_COUNT=$(find "$DIAGRAMS_OUTPUT_PATH" -type f \( -name "*.md" -o -name "*.mmd" \) | wc -l | tr -d ' ')
echo " Total .md/.mmd files: $FILE_COUNT"
else
echo " ⚠️ Directory does not exist!"
fi
echo ""
# The hub-rendered ecosystem.md joins the reference directory only NOW,
# after the engine has run: placed earlier it makes CodeWiki prompt for
# an overwrite (see "Fetch ecosystem facts").
if [ -f /tmp/ecosystem.md ]; then
mkdir -p "$REFERENCE_OUTPUT_PATH"
cp /tmp/ecosystem.md "$REFERENCE_OUTPUT_PATH/ecosystem.md"
echo " Added ecosystem.md to $REFERENCE_OUTPUT_PATH/"
fi
# Stage all .md, .mmd, and .gitignore files from Stage 2 output directories
# CRITICAL: Use git add on full paths to preserve nested directory structure
# This ensures Backend/Authentication/JWT/JWT.md keeps its full path in git
echo "📁 Staging Stage 2 files (excluding JSON/temp files)..."
# Add only .md, .mmd, .gitignore, and allowed JSON files (*-schema.json, *-config.json)
# This excludes CodeWiki intermediate files: module_tree.json, first_module_tree.json, metadata.json
if [ -d "$REFERENCE_OUTPUT_PATH" ]; then
find "$REFERENCE_OUTPUT_PATH" -type f \( \
-name "*.md" -o -name "*.mmd" -o -name ".gitignore" \
-o -name "*-schema.json" -o -name "*-config.json" \
\) -exec git add {} \; 2>/dev/null || true
echo " Added .md/.mmd/.gitignore/*-schema.json/*-config.json from $REFERENCE_OUTPUT_PATH/"
fi
if [ -d "$DIAGRAMS_OUTPUT_PATH" ]; then
find "$DIAGRAMS_OUTPUT_PATH" -type f \( \
-name "*.md" -o -name "*.mmd" -o -name ".gitignore" \
-o -name "*-schema.json" -o -name "*-config.json" \
\) -exec git add {} \; 2>/dev/null || true
echo " Added .md/.mmd/.gitignore/*-schema.json/*-config.json from $DIAGRAMS_OUTPUT_PATH/"
fi
# AGENTS.md carries the hub-rendered ecosystem block ("Fetch ecosystem
# facts" upserts it just before this stage). It sits at the repository
# root, outside every output directory staged above, so it is staged by
# name: the first production run wrote the block and no commit ever
# picked the file up (openframe-cli#383).
if [ -f AGENTS.md ]; then
git add -f AGENTS.md
echo " Added AGENTS.md (ecosystem block)"
fi
# Check if there are changes
STAGED_COUNT=$(git diff --cached --name-only | wc -l)
echo " Staged $STAGED_COUNT files"
# Debug: Show what was actually staged
if [ "$STAGED_COUNT" -gt 0 ]; then
echo ""
echo "🔍 Debug: Files STAGED by git:"
git diff --cached --name-only | head -30
fi
echo ""
if [ "$STAGED_COUNT" -gt 0 ]; then
# Commit and push
git commit -m "docs: Stage 2 - Architecture analysis ($STAGE2_FILES files) [skip ci]"
git push origin "$BRANCH_NAME"
echo "✅ Committed and pushed $STAGED_COUNT files from Stage 2"
set_output "committed" "true"
else
echo "❌ ERROR: No Stage 2 files staged for commit!"
echo " Expected $STAGE2_FILES files but staged 0"
echo " This indicates files were generated but not properly staged"
set_output "committed" "false"
fi
- name: Update PR with Stage 2 Progress
if: always() && steps.commit-stage2.outputs.committed == 'true' && steps.create-initial-pr.outputs.pull-request-number
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }}
STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }}
STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }}
RUN_ID_VAR: ${{ env.RUN_ID }}
REPO_NAME: ${{ github.repository }}
run: |
source /tmp/workflow-helpers.sh
echo "📝 Updating PR description with Stage 2 results..."
# Update PR body
PR_BODY="## 🦩 Flamingo Code Documentation: In Progress
**Run ID:** \`$RUN_ID_VAR\`
**Status:** 🔄 Running...
This PR is being updated as each documentation stage completes.
### Progress
- ✅ Stage 1: Inline Documentation - Completed ($STAGE1_FILES files)
- ✅ Stage 2: Architecture Analysis - Completed ($STAGE2_FILES files)
- ⏳ Stage 3: Tutorial Generation - Running...
- ⏱️ Stage 4: Repository Documentation - Pending
---
🦩 Generated by [Flamingo Code Documentation](https://flamingo.run)"
gh pr edit "$PR_NUMBER" --body "$PR_BODY"
echo "✅ PR #$PR_NUMBER updated with Stage 2 progress"
- name: Report Stage 2 Progress
if: always() && contains(env.STAGES, 'codewiki') && env.HUB_BASE_URL != ''
continue-on-error: true
env:
WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }}
WORKFLOW_RUN_ID: ${{ github.run_id }}
WORKFLOW_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }}
STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }}
# Use outputs from either CodeWiki (stage2) or Claude alternative (stage2_alt)
STAGE2_STATUS: ${{ steps.stage2.outputs.stage2_status || steps.stage2_alt.outputs.stage2_status || 'skipped' }}
STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }}
PR_URL: ${{ steps.create-initial-pr.outputs.pull-request-url }}
PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }}
run: |
source /tmp/workflow-helpers.sh
# Construct callback URL from hub base URL
CALLBACK_URL="${HUB_BASE_URL}/api/doc-orchestrator/webhook"
echo "📤 Reporting Stage 2 (Architecture Analysis) completion with PR URL..."
report_stage_progress "$CALLBACK_URL" "$WEBHOOK_SECRET" "$RUN_ID" "$REPO_ID" \
"$WORKFLOW_RUN_ID" "$WORKFLOW_URL" "tutorials" \
"$STAGE1_STATUS" "$STAGE1_FILES" "$STAGE2_STATUS" "$STAGE2_FILES" "" "0" "" "0" \
"$PR_URL" "$PR_NUMBER"
# =========================================================================
# STAGE 3: AI TUTORIAL GENERATOR (VoltAgent-powered)
# Generate getting started guides and how-to tutorials
# Uses VoltAgent framework with tool-based document generation
# Generates 4 tutorials: user/getting-started, user/common-use-cases,
# dev/getting-started-dev, dev/architecture-overview-dev
# =========================================================================
- name: Install Tutorial Generator Dependencies
# Stage 3 AND Stage 4 run on VoltAgent (generate-repo-docs.cjs requires
# @voltagent/core too). Gated on tutorials alone, a repository configured
# with inline-docs + repo-docs and no tutorials (openframe-saas-mobile,
# openframe-saas-desktop) reached Stage 4 with no dependencies at all:
# "Cannot find module '@voltagent/core'", run 35302244804.
if: contains(env.STAGES, 'tutorials') || contains(env.STAGES, 'repo-docs')
# Isolated deps tree (see Stage 1) - no reconciliation with the target repo, so no
# peer-conflict band-aids. @ai-sdk/anthropic pinned to 2.x (v2 spec): 3.x uses the
# v3 spec, incompatible with VoltAgent's ai@5.x (AI SDK 5).
run: |
mkdir -p "$RUNNER_TEMP/doc-orch-deps" && cd "$RUNNER_TEMP/doc-orch-deps"
npm init -y >/dev/null 2>&1
npm install @voltagent/core@2.9.0 "@ai-sdk/anthropic@2.0.91" zod@3.25.76 glob@13.0.6
- name: Generate Tutorials with VoltAgent
id: stage3
if: contains(env.STAGES, 'tutorials')
env:
# SECURITY: Pass secrets per-step with inline masking
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }}
# Pass through output paths from workflow env (OSS Tenant Structure)
DOCS_OUTPUT_PATH: ${{ env.DOCS_OUTPUT_PATH }}
# Stage 2 outputs (for context)
REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }}
DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }}
# Stage 3 outputs
GETTING_STARTED_OUTPUT_PATH: ${{ env.GETTING_STARTED_OUTPUT_PATH }}
DEVELOPMENT_OUTPUT_PATH: ${{ env.DEVELOPMENT_OUTPUT_PATH }}
# Claude model SSOT — see workflow env CLAUDE_MODEL block
CLAUDE_MODEL: ${{ env.CLAUDE_MODEL }}
# Stage timeout
STAGE3_TIMEOUT_HOURS: ${{ env.STAGE3_TIMEOUT_HOURS }}
# Unified file discovery result (same files as Stage 1 and 2)
SOURCE_FILES_LIST: ${{ steps.discover_files.outputs.source_files_list }}
# NODE_PATH to find modules from /tmp/ scripts
NODE_PATH: ${{ runner.temp }}/doc-orch-deps/node_modules
# YouTube Integration - SECURITY: API key from secrets, NOT dispatch payload
YOUTUBE_ENABLED: ${{ env.YOUTUBE_ENABLED }}
YOUTUBE_CHANNELS: ${{ env.YOUTUBE_CHANNELS }}
YOUTUBE_API_KEY: ${{ secrets.YOUTUBE_API_KEY }}
# Markdown Validation Rules (injected into prompts)
VALIDATION_RULES_PATH: ${{ env.VALIDATION_RULES_PATH }}
# Flamingo Markdown Guidelines (optional)
GUIDELINES_PATH: ${{ env.GUIDELINES_PATH }}
# Stage 3 tracking files (configurable paths)
STAGE3_FILES_TRACKER: ${{ env.STAGE3_FILES_TRACKER }}
STAGE3_STATS_FILE: ${{ env.STAGE3_STATS_FILE }}
# Custom AI Instructions (All Stages)
CUSTOM_REPO_INSTRUCTIONS: ${{ env.CUSTOM_INSTRUCTIONS }}
# Analysis Exclusions
EXCLUDED_PATHS: ${{ env.EXCLUDED_PATHS }}
# External Repositories
EXTERNAL_REPOS: ${{ env.EXTERNAL_REPOS }}
run: |
source /tmp/workflow-helpers.sh
echo "🤖 Stage 3: VoltAgent Tutorial Generator starting..."
# Script already downloaded to /tmp/ in setup step. run_stage records
# the outcome as stage3_status — see its note in workflow-helpers.sh.
run_stage "Stage 3" "$STAGE3_TIMEOUT_HOURS" stage3_status node /tmp/generate-tutorials-voltagent.cjs
# Count files from both OSS Tenant Structure directories
GETTING_STARTED_FILES=$(count_markdown_files "${GETTING_STARTED_OUTPUT_PATH}")
DEVELOPMENT_FILES=$(count_markdown_files "${DEVELOPMENT_OUTPUT_PATH}")
TUTORIAL_FILES=$((GETTING_STARTED_FILES + DEVELOPMENT_FILES))
echo " Getting Started: $GETTING_STARTED_FILES files"
echo " Development: $DEVELOPMENT_FILES files"
echo " Total Stage 3: $TUTORIAL_FILES files"
set_output "stage3_files" "$TUTORIAL_FILES"
# =========================================================================
# COMMIT STAGE 3 RESULTS (Progressive PR)
# =========================================================================
# `!= ''` — the run_stage commit rule, stated once at Stage 1.
- name: Commit and Push Stage 3 Results
if: always() && steps.stage3.outputs.stage3_status != ''
id: commit-stage3
env:
BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }}
STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files }}
GETTING_STARTED_OUTPUT_PATH: ${{ env.GETTING_STARTED_OUTPUT_PATH }}
DEVELOPMENT_OUTPUT_PATH: ${{ env.DEVELOPMENT_OUTPUT_PATH }}
run: |
source /tmp/workflow-helpers.sh
echo "📝 Committing Stage 3 (Tutorial Generation) results..."
# Create .gitignore in VoltAgent output directories
echo "📝 Creating .gitignore files in output directories..."
for output_dir in "$GETTING_STARTED_OUTPUT_PATH" "$DEVELOPMENT_OUTPUT_PATH"; do
if [ -d "$output_dir" ]; then
{
echo "# VoltAgent temp files"
echo "temp/"
echo ""
echo "# JSON intermediate files (except schema/config)"
echo "*.json"
echo "!*-schema.json"
echo "!*-config.json"
} > "$output_dir/.gitignore"
echo " ✅ Created $output_dir/.gitignore"
fi
done
# Stage all .md and .gitignore files from Stage 3 output directories
find "$GETTING_STARTED_OUTPUT_PATH" "$DEVELOPMENT_OUTPUT_PATH" -type f \( -name "*.md" -o -name ".gitignore" \) \
-exec git add -f {} \; 2>/dev/null || true
# Check if there are changes
STAGED_COUNT=$(git diff --cached --name-only | wc -l)
if [ "$STAGED_COUNT" -gt 0 ]; then
# Commit and push
git commit -m "docs: Stage 3 - Tutorial generation ($STAGE3_FILES files) [skip ci]"
git push origin "$BRANCH_NAME"
echo "✅ Committed and pushed $STAGED_COUNT files from Stage 3"
set_output "committed" "true"
else
echo "⚠️ No Stage 3 files to commit"
set_output "committed" "false"
fi
- name: Update PR with Stage 3 Progress
if: always() && steps.commit-stage3.outputs.committed == 'true' && steps.create-initial-pr.outputs.pull-request-number
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }}
STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }}
STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }}
STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || 0 }}
RUN_ID_VAR: ${{ env.RUN_ID }}
REPO_NAME: ${{ github.repository }}
run: |
source /tmp/workflow-helpers.sh
echo "📝 Updating PR description with Stage 3 results..."
# Update PR body
PR_BODY="## 🦩 Flamingo Code Documentation: In Progress
**Run ID:** \`$RUN_ID_VAR\`
**Status:** 🔄 Running...
This PR is being updated as each documentation stage completes.
### Progress
- ✅ Stage 1: Inline Documentation - Completed ($STAGE1_FILES files)
- ✅ Stage 2: Architecture Analysis - Completed ($STAGE2_FILES files)
- ✅ Stage 3: Tutorial Generation - Completed ($STAGE3_FILES files)
- ⏳ Stage 4: Repository Documentation - Running...
---
🦩 Generated by [Flamingo Code Documentation](https://flamingo.run)"
gh pr edit "$PR_NUMBER" --body "$PR_BODY"
echo "✅ PR #$PR_NUMBER updated with Stage 3 progress"
- name: Report Stage 3 Progress
if: always() && contains(env.STAGES, 'tutorials') && env.HUB_BASE_URL != ''
continue-on-error: true
env:
WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }}
WORKFLOW_RUN_ID: ${{ github.run_id }}
WORKFLOW_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }}
STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }}
STAGE2_STATUS: ${{ steps.stage2.outputs.stage2_status || steps.stage2_alt.outputs.stage2_status || 'skipped' }}
STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }}
STAGE3_STATUS: ${{ steps.stage3.outputs.stage3_status || 'skipped' }}
STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || 0 }}
PR_URL: ${{ steps.create-initial-pr.outputs.pull-request-url }}
PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }}
run: |
source /tmp/workflow-helpers.sh
# Construct callback URL from hub base URL
CALLBACK_URL="${HUB_BASE_URL}/api/doc-orchestrator/webhook"
echo "📤 Reporting Stage 3 (AI Tutorial Generator) completion with PR URL..."
report_stage_progress "$CALLBACK_URL" "$WEBHOOK_SECRET" "$RUN_ID" "$REPO_ID" \
"$WORKFLOW_RUN_ID" "$WORKFLOW_URL" "repo-docs" \
"$STAGE1_STATUS" "$STAGE1_FILES" "$STAGE2_STATUS" "$STAGE2_FILES" "$STAGE3_STATUS" "$STAGE3_FILES" "" "0" \
"$PR_URL" "$PR_NUMBER"
# =========================================================================
# STAGE 4: REPOSITORY DOCUMENTATION
# Copies LICENSE.md, SECURITY.md from template repo
# Generates/updates README.md, CONTRIBUTING.md using VoltAgent
# =========================================================================
- name: Generate Repository Documentation
id: stage4
if: contains(env.STAGES, 'repo-docs')
env:
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
TEMPLATE_REPO: ${{ env.TEMPLATE_REPO }}
TEMPLATE_BRANCH: ${{ env.TEMPLATE_BRANCH }}
DOCS_OUTPUT_PATH: ${{ env.DOCS_OUTPUT_PATH }}
# OSS Tenant Structure: All output paths for docs/README.md navigation
REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }}
DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }}
GETTING_STARTED_OUTPUT_PATH: ${{ env.GETTING_STARTED_OUTPUT_PATH }}
DEVELOPMENT_OUTPUT_PATH: ${{ env.DEVELOPMENT_OUTPUT_PATH }}
# Claude model SSOT — see workflow env CLAUDE_MODEL block
CLAUDE_MODEL: ${{ env.CLAUDE_MODEL }}
STAGE4_TIMEOUT_HOURS: ${{ env.STAGE4_TIMEOUT_HOURS }}
# NODE_PATH to find modules from /tmp/ scripts
NODE_PATH: ${{ runner.temp }}/doc-orch-deps/node_modules
# YouTube Integration - SECURITY: API key from secrets, NOT dispatch payload
YOUTUBE_ENABLED: ${{ env.YOUTUBE_ENABLED }}
YOUTUBE_CHANNELS: ${{ env.YOUTUBE_CHANNELS }}
YOUTUBE_API_KEY: ${{ secrets.YOUTUBE_API_KEY }}
# Markdown Validation Rules (injected into prompts)
VALIDATION_RULES_PATH: ${{ env.VALIDATION_RULES_PATH }}
# Flamingo Markdown Guidelines (optional)
GUIDELINES_PATH: ${{ env.GUIDELINES_PATH }}
# Stage 4 tracking files (configurable paths)
STAGE4_FILES_TRACKER: ${{ env.STAGE4_FILES_TRACKER }}
# Custom AI Instructions (All Stages)
CUSTOM_REPO_INSTRUCTIONS: ${{ env.CUSTOM_INSTRUCTIONS }}
# Analysis Exclusions
EXCLUDED_PATHS: ${{ env.EXCLUDED_PATHS }}
# External Repositories
EXTERNAL_REPOS: ${{ env.EXTERNAL_REPOS }}
# README Branding
README_LOGO_DARK: ${{ env.README_LOGO_DARK }}
README_LOGO_LIGHT: ${{ env.README_LOGO_LIGHT }}
README_LOGO_ALT: ${{ env.README_LOGO_ALT }}
run: |
source /tmp/workflow-helpers.sh
echo "📄 Stage 4: Repository Documentation..."
echo " Template: $TEMPLATE_REPO (branch: $TEMPLATE_BRANCH)"
RAW_URL="https://raw.githubusercontent.com/$TEMPLATE_REPO/$TEMPLATE_BRANCH"
# === STEP 1: Copy LICENSE.md and SECURITY.md from template repo ===
echo ""
echo "📥 Fetching LICENSE.md from template repo..."
if curl -fsSL "$RAW_URL/LICENSE.md" -o LICENSE.md 2>/dev/null; then
echo " ✅ LICENSE.md copied"
else
echo " ⚠️ LICENSE.md not found in template repo (non-blocking)"
fi
echo "📥 Fetching SECURITY.md from template repo..."
if curl -fsSL "$RAW_URL/SECURITY.md" -o SECURITY.md 2>/dev/null; then
echo " ✅ SECURITY.md copied"
else
echo " ⚠️ SECURITY.md not found in template repo (non-blocking)"
fi
# === STEP 2: Check existing README ===
echo ""
if [ -f "README.md" ]; then
README_SIZE=$(wc -c < README.md | tr -d ' ')
echo "📝 Found existing README.md ($README_SIZE bytes) - will use as context"
else
echo "📝 No README.md found"
fi
echo " Generating fresh README with OpenFrame branding..."
# === STEP 3: Run VoltAgent script (already downloaded in setup step) ===
echo ""
echo "🤖 Generating repository documentation with VoltAgent..."
# Script already downloaded to /tmp/ in Download Workflow Scripts step
# run_stage records the outcome as stage4_status — see its note in
# workflow-helpers.sh. The file count below is REPORTING, not a
# status: inferring "completed" from it meant a crashed run that left
# a previous commit's README standing reported success.
run_stage "Stage 4" "$STAGE4_TIMEOUT_HOURS" stage4_status node /tmp/generate-repo-docs.cjs
# === STEP 4: Count results ===
echo ""
echo "📊 Stage 4 Results:"
REPO_DOCS=0
for f in README.md CONTRIBUTING.md LICENSE.md SECURITY.md; do
if [ -f "$f" ]; then
SIZE=$(wc -c < "$f" | tr -d ' ')
echo " ✅ $f ($SIZE bytes)"
REPO_DOCS=$((REPO_DOCS + 1))
fi
done
set_output "stage4_files" "$REPO_DOCS"
if [ "$REPO_DOCS" -gt 0 ]; then
echo ""
echo "✅ Stage 4 produced $REPO_DOCS repository documentation file(s)"
else
echo ""
echo "⚠️ Stage 4 produced no repository documentation files"
fi
# =========================================================================
# COMMIT STAGE 4 RESULTS (Progressive PR)
# =========================================================================
# `!= ''` — the run_stage commit rule, stated once at Stage 1.
- name: Commit and Push Stage 4 Results
if: always() && steps.stage4.outputs.stage4_status != ''
id: commit-stage4
env:
BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }}
STAGE4_FILES: ${{ steps.stage4.outputs.stage4_files }}
run: |
source /tmp/workflow-helpers.sh
echo "📝 Committing Stage 4 (Repository Documentation) results..."
# Create .gitignore in Stage 4 managed directories
echo "📝 Creating .gitignore files in managed directories..."
for managed_dir in docs/api docs/deployment docs/operations docs/cli; do
if [ -d "$managed_dir" ]; then
{
echo "# VoltAgent temp files"
echo "temp/"
echo ""
echo "# JSON intermediate files (except schema/config)"
echo "*.json"
echo "!*-schema.json"
echo "!*-config.json"
} > "$managed_dir/.gitignore"
echo " ✅ Created $managed_dir/.gitignore"
fi
done
# Stage repository documentation files
# AGENTS.md is here as the backstop for a run whose Stage 2 commit did not happen.
for f in README.md CONTRIBUTING.md LICENSE.md SECURITY.md AGENTS.md; do
if [ -f "$f" ]; then
git add -f "$f"
fi
done
# Stage Stage 4 managed directories
for managed_dir in docs/api docs/deployment docs/operations docs/cli; do
if [ -d "$managed_dir" ]; then
git add -f "$managed_dir/" 2>/dev/null || true
fi
done
# Stage docs/README.md if exists
if [ -f "docs/README.md" ]; then
git add -f "docs/README.md"
fi
# Check if there are changes
STAGED_COUNT=$(git diff --cached --name-only | wc -l)
if [ "$STAGED_COUNT" -gt 0 ]; then
# Commit and push
git commit -m "docs: Stage 4 - Repository documentation ($STAGE4_FILES files) [skip ci]"
git push origin "$BRANCH_NAME"
echo "✅ Committed and pushed $STAGED_COUNT files from Stage 4"
set_output "committed" "true"
else
echo "⚠️ No Stage 4 files to commit"
set_output "committed" "false"
fi
- name: Update PR with Stage 4 Progress
if: always() && steps.commit-stage4.outputs.committed == 'true' && steps.create-initial-pr.outputs.pull-request-number
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }}
STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }}
STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }}
STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || 0 }}
STAGE4_FILES: ${{ steps.stage4.outputs.stage4_files || 0 }}
RUN_ID_VAR: ${{ env.RUN_ID }}
REPO_NAME: ${{ github.repository }}
run: |
source /tmp/workflow-helpers.sh
echo "📝 Updating PR description with Stage 4 results..."
# Update PR body
PR_BODY="## 🦩 Flamingo Code Documentation: In Progress
**Run ID:** \`$RUN_ID_VAR\`
**Status:** 🔄 Running...
This PR is being updated as each documentation stage completes.
### Progress
- ✅ Stage 1: Inline Documentation - Completed ($STAGE1_FILES files)
- ✅ Stage 2: Architecture Analysis - Completed ($STAGE2_FILES files)
- ✅ Stage 3: Tutorial Generation - Completed ($STAGE3_FILES files)
- ✅ Stage 4: Repository Documentation - Completed ($STAGE4_FILES files)
---
🦩 Generated by [Flamingo Code Documentation](https://flamingo.run)"
gh pr edit "$PR_NUMBER" --body "$PR_BODY"
echo "✅ PR #$PR_NUMBER updated with Stage 4 progress"
# =========================================================================
# VALIDATE GENERATED MARKDOWN
# Warn-only validation (never blocks PRs)
# =========================================================================
- name: Validate Generated Markdown
if: always()
continue-on-error: true # NEVER block PR - validation is warn-only
env:
DOCS_OUTPUT_DIR: ${{ env.DOCS_OUTPUT_PATH }}
# OSS Tenant Structure paths for validation
REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }}
DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }}
GETTING_STARTED_OUTPUT_PATH: ${{ env.GETTING_STARTED_OUTPUT_PATH }}
DEVELOPMENT_OUTPUT_PATH: ${{ env.DEVELOPMENT_OUTPUT_PATH }}
NODE_PATH: ${{ runner.temp }}/doc-orch-deps/node_modules
run: |
echo "📋 Validating generated markdown against Flamingo guidelines..."
node /tmp/validate-markdown.js "$DOCS_OUTPUT_DIR" 2>&1 || true
# Validate OSS Tenant Structure outputs
node /tmp/validate-markdown.js "$REFERENCE_OUTPUT_PATH" 2>&1 || true
node /tmp/validate-markdown.js "$GETTING_STARTED_OUTPUT_PATH" 2>&1 || true
node /tmp/validate-markdown.js "$DEVELOPMENT_OUTPUT_PATH" 2>&1 || true
echo "✅ Validation complete (warnings logged above, non-blocking)"
# Report Stage 4 Progress
- name: Report Stage 4 Progress
if: always() && env.HUB_BASE_URL != ''
continue-on-error: true
env:
WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }}
WORKFLOW_RUN_ID: ${{ github.run_id }}
WORKFLOW_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }}
STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }}
STAGE2_STATUS: ${{ steps.stage2.outputs.stage2_status || steps.stage2_alt.outputs.stage2_status || 'skipped' }}
STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }}
STAGE3_STATUS: ${{ steps.stage3.outputs.stage3_status || 'skipped' }}
STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || 0 }}
STAGE4_STATUS: ${{ steps.stage4.outputs.stage4_status || 'skipped' }}
STAGE4_FILES: ${{ steps.stage4.outputs.stage4_files || 0 }}
PR_URL: ${{ steps.create-initial-pr.outputs.pull-request-url }}
PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }}
run: |
source /tmp/workflow-helpers.sh
# Construct callback URL from hub base URL
CALLBACK_URL="${HUB_BASE_URL}/api/doc-orchestrator/webhook"
echo "📤 Reporting Stage 4 (Repository Documentation) completion with PR URL..."
report_stage_progress "$CALLBACK_URL" "$WEBHOOK_SECRET" "$RUN_ID" "$REPO_ID" \
"$WORKFLOW_RUN_ID" "$WORKFLOW_URL" "creating-pr" \
"$STAGE1_STATUS" "$STAGE1_FILES" "$STAGE2_STATUS" "$STAGE2_FILES" "$STAGE3_STATUS" "$STAGE3_FILES" \
"$STAGE4_STATUS" "$STAGE4_FILES" "$PR_URL" "$PR_NUMBER"
# =========================================================================
# CREATE PULL REQUEST
# =========================================================================
- name: Cleanup Temporary Files
run: |
source /tmp/workflow-helpers.sh
echo "🧹 Cleaning up temporary files before PR creation..."
# Remove stats files
cleanup_path ".doc-stage1-stats.json"
cleanup_path ".doc-stage3-stats.json"
# Remove run status file (created for the initial PR).
# Legacy name kept so branches started before the rename still clean up.
cleanup_path ".flamingo-ai-technical-writer-status.md"
cleanup_path ".doc-pipeline-status.md"
# Note: .doc-orchestrator-source-files.txt is now in /tmp/ (auto-cleanup)
# Remove npm artifacts (installed for scripts)
cleanup_path "node_modules"
cleanup_path "package.json"
cleanup_path "package-lock.json"
# NOTE: /tmp/workflow-helpers.sh is cleaned up in final webhook step
echo "✅ Cleanup complete"
- name: Stage Generated Documentation
id: stage-docs
run: |
source /tmp/workflow-helpers.sh
echo "📁 Staging generated documentation files..."
echo " DOCS_OUTPUT_PATH: $DOCS_OUTPUT_PATH"
echo " OSS Tenant Structure paths:"
echo " Stage 2: $REFERENCE_OUTPUT_PATH (reference)"
echo " Stage 2: $DIAGRAMS_OUTPUT_PATH (diagrams)"
echo " Stage 3: $GETTING_STARTED_OUTPUT_PATH (getting-started)"
echo " Stage 3: $DEVELOPMENT_OUTPUT_PATH (development)"
# Count untracked/modified files before staging
BEFORE_COUNT=$(git status --porcelain | wc -l)
echo " Total changed files: $BEFORE_COUNT"
# Stage ALL .md and .mmd files anywhere in the repo (for inline docs generated next to source files)
# This catches Stage 1 inline docs (hidden: .FileName.md), Stage 2 reference/diagrams, Stage 3 tutorials, and Stage 4 repo docs
echo " Finding all .md and .mmd files to stage (including hidden and diagrams)..."
# Find all .md and .mmd files recursively, including hidden files (.*.md)
# Includes README.md, CONTRIBUTING.md, LICENSE.md, SECURITY.md from Stage 4
# Includes .mmd Mermaid diagram files from Stage 2 (CodeWiki/Claude architecture)
find . \( -name "*.md" -o -name ".*.md" -o -name "*.mmd" \) -type f \
-not -path "./node_modules/*" \
-not -path "./.git/*" \
-not -name "CHANGELOG.md" \
-exec git add -f {} \; 2>/dev/null || true
# Show what .md and .mmd files exist (for debugging)
echo ""
echo "📋 All .md and .mmd files found (including hidden inline docs and diagrams):"
find . \( -name "*.md" -o -name ".*.md" -o -name "*.mmd" \) -type f \
-not -path "./node_modules/*" \
-not -path "./.git/*" \
-not -name "CHANGELOG.md" | head -100
# Count staged files
STAGED_COUNT=$(git diff --cached --name-only | wc -l)
echo " Staged files: $STAGED_COUNT"
set_output "staged_count" "$STAGED_COUNT"
# Show what was staged
echo ""
echo "📋 Staged files:"
git diff --cached --name-only | head -50
# "Changes" means the BRANCH differs from the documented source head, not
# that this final sweep found something left to stage: every stage commits
# its own output as it goes, so a run whose stages all committed (inline
# docs, README) left nothing here and was reported as no_changes while its
# pull request held twenty files (openframe-saas-mobile run 35302244804).
# The status file is the run's own bookkeeping, never a documentation change.
if [ "$STAGED_COUNT" -eq "0" ] && git diff --quiet "$SOURCE_HEAD_SHA" HEAD -- . ':!.flamingo-ai-technical-writer-status.md'; then
echo ""
echo "⚠️ No documentation files to stage, and the branch holds no documentation change"
set_output "has_changes" "false"
else
set_output "has_changes" "true"
fi
- name: Update Final PR Status
if: steps.create-initial-pr.outputs.pull-request-number
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }}
STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }}
STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }}
STAGE2_STATUS: ${{ steps.stage2.outputs.stage2_status || steps.stage2_alt.outputs.stage2_status || 'skipped' }}
STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }}
STAGE3_STATUS: ${{ steps.stage3.outputs.stage3_status || 'skipped' }}
STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || 0 }}
STAGE4_STATUS: ${{ steps.stage4.outputs.stage4_status || 'skipped' }}
STAGE4_FILES: ${{ steps.stage4.outputs.stage4_files || 0 }}
RUN_ID_VAR: ${{ env.RUN_ID }}
REPO_NAME: ${{ github.repository }}
run: |
source /tmp/workflow-helpers.sh
echo "📝 Updating PR with final status..."
# Remove "in-progress" label
gh pr edit "$PR_NUMBER" --remove-label "in-progress" || true
# Update title to remove [IN PROGRESS]
gh pr edit "$PR_NUMBER" --title "🦩 Flamingo Code Documentation"
# Update body with final results
PR_BODY="## 🦩 Flamingo Code Documentation: Complete
**Run ID:** \`$RUN_ID_VAR\`
### Stage 1: Inline Documentation
- Status: $STAGE1_STATUS
- Files generated: $STAGE1_FILES
- Generated .md files next to source classes explaining their purpose
### Stage 2: Architecture Analysis
- Status: $STAGE2_STATUS
- Files generated: $STAGE2_FILES
- Architecture overview and module documentation
### Stage 3: AI Tutorial Generator
- Status: $STAGE3_STATUS
- Files generated: $STAGE3_FILES
- Getting started guides and how-to tutorials
### Stage 4: Repository Documentation
- Status: $STAGE4_STATUS
- Files generated: $STAGE4_FILES
- README.md, CONTRIBUTING.md, LICENSE.md, SECURITY.md
---
**Review checklist:**
- [ ] Check generated inline docs for accuracy
- [ ] Review architecture documentation
- [ ] Test code examples in tutorials
- [ ] Review README.md and CONTRIBUTING.md updates
---
🦩 Generated by [Flamingo Code Documentation](https://flamingo.run)"
gh pr edit "$PR_NUMBER" --body "$PR_BODY"
echo "✅ PR #$PR_NUMBER updated with final status"
# =========================================================================
# SEND WEBHOOK NOTIFICATION
# =========================================================================
- name: Send Webhook Notification
if: always() && env.HUB_BASE_URL != ''
continue-on-error: true # Don't fail the workflow if callback fails
env:
# SECURITY: Pass secret per-step with inline masking
WEBHOOK_SECRET: ${{ secrets.DOC_ORCH_WEBHOOK_SECRET }}
WORKFLOW_RUN_ID: ${{ github.run_id }}
WORKFLOW_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
HAS_CHANGES: ${{ steps.stage-docs.outputs.has_changes }}
JOB_STATUS: ${{ job.status }}
PR_URL: ${{ steps.create-initial-pr.outputs.pull-request-url }}
PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number || 'null' }}
SAFE_RUN_ID: ${{ steps.branch-name-early.outputs.safe_run_id }}
# Single source of truth for the branch name (was rebuilt by hand below,
# which silently drifted from "Create PR Branch" on every rename)
BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }}
STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }}
STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }}
# Stage 2: Check both CodeWiki and Claude alternative, mark as failed if step failed
STAGE2_STATUS: ${{ steps.stage2.outcome == 'failure' && 'failed' || steps.stage2_alt.outcome == 'failure' && 'failed' || steps.stage2.outputs.stage2_status || steps.stage2_alt.outputs.stage2_status || 'skipped' }}
STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }}
STAGE3_STATUS: ${{ steps.stage3.outputs.stage3_status || 'skipped' }}
STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || 0 }}
STAGE4_STATUS: ${{ steps.stage4.outputs.stage4_status || 'skipped' }}
STAGE4_FILES: ${{ steps.stage4.outputs.stage4_files || 0 }}
# Track if critical steps failed (continue-on-error: false steps)
STAGE2_OUTCOME: ${{ steps.stage2.outcome || 'skipped' }}
CODEWIKI_INSTALL_OUTCOME: ${{ steps.codewiki_install.outcome || 'skipped' }}
CODEWIKI_CONFIG_OUTCOME: ${{ steps.codewiki_config.outcome || 'skipped' }}
run: |
# Bootstrap-failure fallback (shared failure-net standard with the
# code-review workflow): if workflow-helpers.sh never downloaded, no
# helper exists to report the failure — a minimal guarded curl posts
# it so the hub's run row fails NOW instead of waiting for the reaper.
if [ ! -f /tmp/workflow-helpers.sh ]; then
echo "::error::workflow-helpers.sh missing — sending bootstrap-failure callback"
# The bearer goes through a 0600 config file, never argv — see
# curlAuthPreamble in lib/config/workflow-scripts-bootstrap.ts.
CURL_CFG=$(mktemp) && chmod 600 "$CURL_CFG"
trap 'rm -f "$CURL_CFG"' EXIT
printf 'header = "Authorization: Bearer %s"\n' "$WEBHOOK_SECRET" > "$CURL_CFG"
curl -sS --max-time 30 -K "$CURL_CFG" -X POST "${HUB_BASE_URL}/api/doc-orchestrator/webhook" \
-H "Content-Type: application/json" \
-d "{\"run_id\":\"$RUN_ID\",\"repo_id\":\"$REPO_ID\",\"status\":\"failure\",\"workflow_run_id\":$WORKFLOW_RUN_ID,\"workflow_url\":\"$WORKFLOW_URL\",\"error\":\"Script bootstrap failed: workflow-helpers.sh never downloaded from the hub.\"}" || true
exit 1
fi
source /tmp/workflow-helpers.sh
# Construct callback URL from hub base URL
CALLBACK_URL="${HUB_BASE_URL}/api/doc-orchestrator/webhook"
echo "📊 Determining final workflow status..."
echo " JOB_STATUS: $JOB_STATUS"
echo " HAS_CHANGES: $HAS_CHANGES"
echo " STAGE2_OUTCOME: $STAGE2_OUTCOME"
echo " CODEWIKI_INSTALL_OUTCOME: $CODEWIKI_INSTALL_OUTCOME"
echo " CODEWIKI_CONFIG_OUTCOME: $CODEWIKI_CONFIG_OUTCOME"
# CRITICAL: Determine final status - NEVER return "running"
# Default to failure, only set success if everything checks out
STATUS="failure"
# Check for cancelled job first
if [ "$JOB_STATUS" = "cancelled" ]; then
STATUS="cancelled"
echo " ❌ Status: cancelled (workflow was cancelled)"
# Check if critical stage 2 (CodeWiki) failed - this has continue-on-error: false
elif [ "$STAGE2_OUTCOME" = "failure" ]; then
STATUS="failure"
echo " ❌ Status: failure (CodeWiki stage failed)"
# Check if CodeWiki installation failed
elif [ "$CODEWIKI_INSTALL_OUTCOME" = "failure" ]; then
STATUS="failure"
echo " ❌ Status: failure (CodeWiki installation failed)"
# Check if CodeWiki configuration failed
elif [ "$CODEWIKI_CONFIG_OUTCOME" = "failure" ]; then
STATUS="failure"
echo " ❌ Status: failure (CodeWiki configuration failed)"
# Check overall job status
elif [ "$JOB_STATUS" != "success" ]; then
STATUS="failure"
echo " ❌ Status: failure (job status: $JOB_STATUS)"
# Check if we have any documentation changes
elif [ "$HAS_CHANGES" != "true" ]; then
STATUS="no_changes"
echo " ⚠️ Status: no_changes (no documentation files generated)"
else
STATUS="success"
echo " ✅ Status: success"
fi
# SAFETY CHECK: Ensure status is NEVER "running"
if [ "$STATUS" = "running" ] || [ -z "$STATUS" ]; then
echo " 🚨 SAFETY: Detected invalid status '$STATUS', forcing to 'failure'"
STATUS="failure"
fi
echo ""
echo "📤 Final status to report: $STATUS"
# Branch actually created by "Create PR Branch". Falls back to the same
# formula only when that step never ran (this step is `if: always()`).
SAFE_BRANCH="${BRANCH_NAME:-docs/flamingo-ai-technical-writer-$SAFE_RUN_ID}"
# Send final webhook using helper function
report_final_status "$CALLBACK_URL" "$WEBHOOK_SECRET" "$RUN_ID" "$REPO_ID" \
"$WORKFLOW_RUN_ID" "$WORKFLOW_URL" "$STATUS" "$PR_URL" "$PR_NUMBER" "$SAFE_BRANCH" \
"$STAGE1_STATUS" "$STAGE1_FILES" "$STAGE2_STATUS" "$STAGE2_FILES" "$STAGE3_STATUS" "$STAGE3_FILES" \
"$STAGE4_STATUS" "$STAGE4_FILES"
# Final cleanup: remove workflow helpers file
cleanup_path "/tmp/workflow-helpers.sh"
- name: Pipeline Summary
if: always()
run: |
echo "## 🦩 Flamingo Code Documentation Summary" >> $GITHUB_STEP_SUMMARY
echo "" >> $GITHUB_STEP_SUMMARY
echo "| Stage | Status | Files |" >> $GITHUB_STEP_SUMMARY
echo "|-------|--------|-------|" >> $GITHUB_STEP_SUMMARY
echo "| Inline Docs | ${{ steps.stage1.outputs.stage1_status || 'skipped' }} | ${{ steps.stage1.outputs.stage1_files || '0' }} |" >> $GITHUB_STEP_SUMMARY
echo "| Architecture | ${{ steps.stage2.outputs.stage2_status || steps.stage2_alt.outputs.stage2_status || 'skipped' }} | ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || '0' }} |" >> $GITHUB_STEP_SUMMARY
echo "| Tutorials | ${{ steps.stage3.outputs.stage3_status || 'skipped' }} | ${{ steps.stage3.outputs.stage3_files || '0' }} |" >> $GITHUB_STEP_SUMMARY
echo "| Repo Docs | ${{ steps.stage4.outputs.stage4_status || 'skipped' }} | ${{ steps.stage4.outputs.stage4_files || '0' }} |" >> $GITHUB_STEP_SUMMARY
echo "" >> $GITHUB_STEP_SUMMARY
if [ -n "${{ steps.create-initial-pr.outputs.pull-request-url }}" ]; then
echo "**Pull Request:** ${{ steps.create-initial-pr.outputs.pull-request-url }}" >> $GITHUB_STEP_SUMMARY
fi