Skip to content

feat(codegen): CU-17tkuw5tebe publish release-bound osquery schemas #91

feat(codegen): CU-17tkuw5tebe publish release-bound osquery schemas

feat(codegen): CU-17tkuw5tebe publish release-bound osquery schemas #91

# Flamingo Code Documentation
# =============================================================================
# Installed into a target repository by the multi-platform hub ("Setup
# Workflow"). The hub dispatches it to generate this repository's
# documentation and open a pull request with the result.
#
# Jobs
# code-graph Deterministic code graph (no model call). Runs on pushes to
# the default branch, on the hub's `flamingo-code-graph`
# re-dispatch, and on a manual run with graph_only=true.
# doc-pipeline The documentation run, dispatched by the hub:
# Stage 0 code graph (same build as the code-graph job)
# Stage 1 inline docs, one hidden .md beside each source file
# Stage 2 reference docs: CodeWiki, or the Claude
# architecture analysis where CodeWiki cannot parse
# the primary language
# Stage 3 tutorials (getting started, development)
# Stage 4 repository docs: README, CONTRIBUTING, docs index
#
# Stages 2 (Claude), 3 and 4 follow the code reviewer's agentic pattern
# (templates/scripts/code-documentation-lib.mjs): the hub serves this
# repository's settings, the script packs the material, ONE hub call writes
# ONE document (forced `emit_document`; the hub's read tools once the
# repository's visibility is resolved), a deterministic gate checks every
# repository path it names, and the script, never the model, picks where each
# document is written.
#
# Fleet contracts (renaming any of these needs a fleet-wide re-push): the
# installed path .github/workflows/code-documentation.yml (the setup PR
# hard-deletes the old doc-orchestrator.yml), the repository_dispatch type
# `code-documentation`, the secrets below, and the script file names the hub serves.
#
# Repository secrets (Settings > Secrets and variables > Actions)
# FLAMINGO_HUB_SECRET Required. Authenticates every hub call.
# ANTHROPIC_API_KEY CodeWiki (stage 2) only; every other stage calls
# Claude through the hub.
# OPENAI_API_KEY CodeWiki (stage 2).
# YOUTUBE_API_KEY Optional. YouTube embeds in stages 3 and 4.
name: 🦩 Flamingo Code Documentation
on:
# Push trigger — two things ride it. It registers the workflow with GitHub
# Actions (required for the workflow_dispatch API), and on the repository's
# DEFAULT branch it runs the `code-graph` job below, which re-indexes the
# code graph the hub serves to the code reviewer and to the documentation
# stages. The documentation pipeline itself NEVER runs on push (see its
# `if:`). Documentation and markdown are ignored on purpose: a docs PR
# merging must not rebuild a graph that only source files can change.
push:
paths-ignore:
- 'docs/**'
- '**.md'
# Multi-repo change sets, LIVE: every pull request event (a `Depends-On:` line
# added, a push, a merge) asks the hub to refresh its change sets at once, and
# the same run builds the PR's overlay when the hub says it is a member whose
# overlay is missing or stale. ONLY the code-graph job runs on this event (see
# both jobs' `if:`); no path filter, because a description edit is the event.
pull_request:
types: [opened, reopened, edited, synchronize, ready_for_review, closed]
repository_dispatch:
# `code-documentation` (CODE_DOCUMENTATION_DISPATCH_EVENT_TYPE) runs the
# documentation pipeline. `flamingo-code-graph`
# (CODE_GRAPH_DISPATCH_EVENT_TYPE in lib/config/code-graph-workflow.ts)
# runs ONLY the graph job; the hub's reconcile job sends it when a
# repository's graph is missing or stale.
types: [code-documentation, flamingo-code-graph]
workflow_dispatch:
# ═══════════════════════════════════════════════════════════════════════════
# GENERATED FROM SINGLE SOURCE OF TRUTH: lib/config/code-documentation-params.ts
# This section is auto-generated at runtime when creating workflow PRs
# ═══════════════════════════════════════════════════════════════════════════
inputs:
# Individual parameters
run_id:
description: 'Unique execution ID'
required: true
repo_id:
description: 'Repository ID from database'
required: true
hub_base_url:
description: 'Hub base URL (e.g., https://product-hub.flamingo.so)'
required: true
stages:
description: 'Pipeline stages to execute'
required: true
dependencies:
description: 'Comma-separated dependency repos'
required: false
default: ''
source_branch:
description: 'Branch to analyze code from'
required: true
source_files_limit:
description: 'Max source files to process (0 = unlimited)'
required: true
claude_model:
description: 'Claude model ID for Stage 1/3/4 + Stage 2 Claude-arch fallback (SSOT from hub)'
required: true
codewiki_config:
description: 'Complete CodeWiki configuration (per-phase models, engine, stages, depth)'
required: true
output_paths:
description: 'Output paths configuration'
required: true
timeout:
description: 'Timeout in hours'
required: true
youtube_config:
description: 'YouTube integration configuration (channels only - API key in secrets)'
required: true
readme_config:
description: 'README logo configuration'
required: true
custom_repo_instructions:
description: 'Custom AI instructions'
required: false
default: ''
external_repos:
description: 'External repos JSON'
required: false
default: '[]'
stage_count:
description: 'Total number of pipeline stages'
required: false
default: '4'
graph_only:
description: 'true = run only the code-graph job (no documentation run)'
required: false
default: 'false'
# ═══════════════════════════════════════════════════════════════════════════
# END GENERATED SECTION
# ═══════════════════════════════════════════════════════════════════════════
env:
# SECURITY: Only NON-SENSITIVE variables in job-level env
# Secrets are passed per-step to avoid exposure in job setup logs
# Run configuration (non-sensitive)
RUN_ID: ${{ github.event.client_payload.run_id || github.event.inputs.run_id || github.run_id }}
REPO_ID: ${{ github.event.client_payload.repo_id || github.event.inputs.repo_id || '' }}
# A push event carries no payload, so the graph job falls back to the org
# Actions variable — the same fallback the code-review workflow uses.
HUB_BASE_URL: ${{ github.event.client_payload.hub_base_url || github.event.inputs.hub_base_url || vars.FLAMINGO_HUB_BASE_URL || '' }}
# 'true' = run only the code-graph job (a manual workflow_dispatch; the hub's
# own documentation dispatches always send 'false').
GRAPH_ONLY: ${{ github.event.client_payload.graph_only || github.event.inputs.graph_only || 'false' }}
# No literal fallback: the stage list lives in CODE_DOCUMENTATION_STAGES on the
# hub and is lifted into the payload per repo. A literal here would silently
# restore all four stages on a payload gap, and "Remove docs this run
# regenerates" would already have wiped the docs tree: the run would delete
# docs and regenerate nothing.
STAGES: ${{ github.event.client_payload.stages || github.event.inputs.stages || '' }}
DEPENDENCIES: ${{ github.event.client_payload.dependencies || github.event.inputs.dependencies || '' }}
STAGE_COUNT: ${{ github.event.client_payload.stage_count || github.event.inputs.stage_count || '4' }}
# Branch to checkout for code analysis (github_branch from repo config)
SOURCE_BRANCH: ${{ github.event.client_payload.source_branch || github.event.inputs.source_branch || 'main' }}
# Debug/testing: limit total source files to analyze (0=unlimited)
# Files beyond this limit are DELETED - all stages then process remaining files
SOURCE_FILES_LIMIT: ${{ github.event.client_payload.source_files_limit || github.event.inputs.source_files_limit || '0' }}
# Claude model SSOT: the hub's CODE_DOCUMENTATION_DEFAULT_MODEL
# (lib/constants/ai-models.ts). The stage 1, 3 and 4 generators and the
# stage 2 Claude analysis all read this variable; no shipped script holds a
# literal model id, and every one throws if CLAUDE_MODEL is empty. Re-run
# "Setup Workflow" after bumping the hub-side constant to propagate it.
# There is deliberately NO companion request-shape variable: every stage
# except CodeWiki calls Claude through the hub (/api/ci/claude), which
# resolves the model's request shape itself.
CLAUDE_MODEL: ${{ github.event.client_payload.claude_model || github.event.inputs.claude_model || '' }}
# =============================================================================
# JSON-grouped parameters to stay under GitHub Actions 25-parameter limit
# These are parsed early in the workflow to extract individual values
# =============================================================================
# NOTE: these fall back to EMPTY, not '{}'. An empty-object default made the
# `[ -z ... ]` presence checks below unreachable, so a missing payload silently
# produced `null` for every jq lookup and propagated as `--cluster-model null`.
# The hub always sends a complete, deep-merged blob (buildPayloadFromRepo).
CODEWIKI_CONFIG_JSON: ${{ github.event.client_payload.codewiki_config || github.event.inputs.codewiki_config || '' }}
OUTPUT_PATHS_JSON: ${{ github.event.client_payload.output_paths || github.event.inputs.output_paths || '' }}
# Stage timeouts (in hours) — single source of truth, downstream steps reference
# `env.STAGE_TIMEOUT_HOURS` directly. Stage 4 is the exception (1h vs 24h cap).
STAGE_TIMEOUT_HOURS: ${{ github.event.client_payload.timeout || github.event.inputs.timeout || '24' }}
# Aliases preserved for downstream step env: keys (they reference these names
# by string). All resolve to the same single source.
STAGE1_TIMEOUT_HOURS: ${{ github.event.client_payload.timeout || github.event.inputs.timeout || '24' }}
# Stage 1 incremental push: commit + push the PR branch every N generated inline
# docs. A 24h stage that gets cancelled used to lose ALL of its work because the
# only commit happened after the generator returned. Bound the loss to N files.
STAGE1_PUSH_INTERVAL: ${{ github.event.client_payload.stage1_push_interval || '100' }}
STAGE2_TIMEOUT_HOURS: ${{ github.event.client_payload.timeout || github.event.inputs.timeout || '24' }}
STAGE3_TIMEOUT_HOURS: ${{ github.event.client_payload.timeout || github.event.inputs.timeout || '24' }}
# Stage 4: Repository Documentation
TEMPLATE_REPO: ${{ github.event.client_payload.template_repo || 'flamingo-stack/openframe-oss-tenant' }}
TEMPLATE_BRANCH: ${{ github.event.client_payload.template_branch || 'main' }}
STAGE4_TIMEOUT_HOURS: ${{ github.event.client_payload.stage4_timeout || '1' }}
# YouTube Integration (Stage 3 + Stage 4) - JSONB configuration (API key from secrets)
YOUTUBE_CONFIG_JSON: ${{ github.event.client_payload.youtube_config || github.event.inputs.youtube_config || '' }}
# README Configuration - JSONB configuration for logo branding
README_CONFIG_JSON: ${{ github.event.client_payload.readme_config || github.event.inputs.readme_config || '' }}
# Custom AI Instructions (All Stages) - Repository-specific instructions for AI generation
CUSTOM_REPO_INSTRUCTIONS: ${{ github.event.client_payload.custom_repo_instructions || github.event.inputs.custom_repo_instructions || '' }}
# External Repositories - JSON array of external repo configurations
EXTERNAL_REPOS: ${{ github.event.client_payload.external_repos || github.event.inputs.external_repos || '[]' }}
# ========================================================================
# Repository Context - CRITICAL for preventing AI URL hallucinations
# These values are passed to ALL AI prompts to ensure correct GitHub URLs
# ========================================================================
GITHUB_REPOSITORY: ${{ github.repository }} # e.g., "flamingo-stack/openframe-oss-tenant"
GITHUB_REPOSITORY_OWNER: ${{ github.repository_owner }} # e.g., "flamingo-stack"
GITHUB_SERVER_URL: ${{ github.server_url }} # e.g., "https://github.com"
# Analysis Exclusions - Complete array of glob patterns to exclude from repository analysis
# (build artifacts, dependencies, the hub's own checkout, cloned dependency repos)
EXCLUDED_PATHS: '**/node_modules/**,**/.git/**,**/target/**,**/dist/**,**/build/**,**/.next/**,**/out/**,**/coverage/**,**/vendor/**,**/.yalc/**,**/.turbo/**,**/.gradle/**,**/__pycache__/**,**/.terraform/**,**/.venv/**,**/venv/**,**/multi-platform-hub/**,**/deps-*/**'
# The documentation pull request's branch prefix (CODE_DOCUMENTATION_BRANCH_PREFIX, written by npm run workflows:sync)
DOCS_BRANCH_PREFIX: 'code-documentation/'
README_LOGO_ALT: 'OpenFrame Logo'
jobs:
# ===========================================================================
# CODE GRAPH: deterministic, no model call. Tags every public symbol,
# import and manifest of the checkout (code-graph-build.mjs) and uploads the
# result to the hub, which promotes a default-branch snapshot to `live` and
# serves it to the code reviewer (consumers of a symbol a PR removes) and to
# the documentation stages (the derived ecosystem.md). Runs on every push to
# the default branch, on the hub's `flamingo-code-graph` re-dispatch, and on
# a manual workflow_dispatch with graph_only=true. There is no webhook
# callback: the upload IS the report. The same build also runs as stage 0 of
# a full documentation run (inside doc-pipeline, below).
# ===========================================================================
code-graph:
runs-on: ubuntu-latest
timeout-minutes: 20
permissions:
contents: read
# THREE groups, so no two kinds of run ever cancel each other:
# - a `pull_request` event runs in `-pr-<n>`, and NEVER cancels a running one
# (`cancel-in-progress` is false for it): a newer event QUEUES behind the run
# in progress (GitHub keeps one pending run per group, replacing an older
# pending one, so the newest event still runs). Cancelling would kill an
# overlay build the hub already recorded as this PR's own (a description
# edit or the merge seconds after a push), and nothing would rebuild that
# head until the retry window passed;
# - an OVERLAY build the hub's `code-change-sets` job dispatches
# (overlay_pr / overlay_head_sha in the payload) runs in `-overlay-<n>`: a
# newer dispatch for the same PR supersedes the older one;
# - everything else (the default branch's rebuild) keeps the repository group.
# A dispatched build and the PR's own run may build the same head at once;
# that is harmless: the hub upserts an overlay on (repo, pr, head, generator).
# A bot's own description edit (the hub writing its block) is excluded by the
# `if:` below, but a job may join its concurrency group before that `if:` is
# evaluated: it gets a group of its own (`-bot-<run id>`), so it can never
# cancel the PR's run that is building the overlay the same refresh asked for.
concurrency:
group: flamingo-code-graph-${{ github.repository }}${{ github.event.client_payload.overlay_pr && format('-overlay-{0}', github.event.client_payload.overlay_pr) || github.event.pull_request.number && format('-pr-{0}', github.event.pull_request.number) || '' }}${{ (github.event.action == 'edited' && github.event.sender.type == 'Bot') && format('-bot-{0}', github.run_id) || '' }}
cancel-in-progress: ${{ github.event_name != 'pull_request' }}
if: >-
(github.event_name == 'push' && github.ref == format('refs/heads/{0}', github.event.repository.default_branch))
|| github.event.action == 'flamingo-code-graph'
|| (github.event_name == 'workflow_dispatch' && github.event.inputs.graph_only == 'true')
|| (github.event_name == 'pull_request'
&& github.event.pull_request.head.repo.full_name == github.repository
&& !startsWith(github.event.pull_request.head.ref, 'ai-fix/')
&& !startsWith(github.event.pull_request.head.ref, 'setup/')
&& !startsWith(github.event.pull_request.head.ref, 'code-documentation/')
&& !(github.event.action == 'edited' && github.event.sender.type == 'Bot'))
# The overlay coordinates, spelled ONCE: the hub's dispatch payload, or this PR's own event (the steps
# below build only after the live refresh answered `build`). Both empty on a default-branch rebuild.
env:
OVERLAY_PR: ${{ github.event.client_payload.overlay_pr || github.event.pull_request.number || '' }}
OVERLAY_HEAD_SHA: ${{ github.event.client_payload.overlay_head_sha || github.event.pull_request.head.sha || '' }}
steps:
# Fail LOUD, not silent: a push on a repo whose org never set
# FLAMINGO_HUB_BASE_URL would otherwise curl an empty origin and die with
# an unrelated error. Also normalizes a trailing slash ONCE.
- name: Validate configuration
id: config
# A pull_request event must never turn a PR check red: the refresh is advisory
# (the hub's webhook path and schedule are the net), so a missing hub URL or
# secret on a PR only skips it.
continue-on-error: ${{ github.event_name == 'pull_request' }}
run: |
if [ -z "$HUB_BASE_URL" ]; then
echo "::error title=Hub URL missing::HUB_BASE_URL is empty. Set the organization Actions variable FLAMINGO_HUB_BASE_URL, or pass hub_base_url in the dispatch payload."
exit 1
fi
echo "HUB_BASE_URL=${HUB_BASE_URL%/}" >> "$GITHUB_ENV"
echo "Hub: ${HUB_BASE_URL%/}"
# The shared script bootstrap (byte-mirrored from workflow-scripts-bootstrap.ts,
# asserted by the build gate). BOTH graph scripts are downloaded: the builder
# imports ./code-graph-lib.mjs from its own directory.
- name: Download graph scripts
id: scripts
if: github.event_name != 'pull_request' || steps.config.outcome == 'success'
continue-on-error: ${{ github.event_name == 'pull_request' }}
env:
WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }}
run: |
# Function to download and verify script
SCRIPT_MANIFEST=/tmp/flamingo-script-manifest.json
# WEBHOOK_SECRET reaches curl through a 0600 config file, never argv — see
# curlAuthPreamble, which always traps the removal.
CURL_CFG=$(mktemp) && chmod 600 "$CURL_CFG"
trap 'rm -f "$CURL_CFG"' EXIT
printf 'header = "Authorization: Bearer %s"\n' "$WEBHOOK_SECRET" > "$CURL_CFG"
# The scripts surface. load_script_manifest pins SCRIPTS_BASE_URL to it.
CI_SCRIPTS_URL="${HUB_BASE_URL%/}/api/ci/scripts"
# _try_manifest <base> — 0 loaded, 1 no manifest surface there, 2 fatal.
# The manifest is asked for ONE group: its keys are the files to download.
_try_manifest() {
local base="$1" code
code=$(curl -sS -w '%{http_code}' -o "$SCRIPT_MANIFEST" \
-K "$CURL_CFG" \
"$base/manifest.json?group=$SCRIPT_GROUP") || code="000"
if [ "$code" = "404" ]; then rm -f "$SCRIPT_MANIFEST"; return 1; fi
if [ "$code" != "200" ]; then
echo "❌ manifest request to $base failed (HTTP $code)"
rm -f "$SCRIPT_MANIFEST"
return 2
fi
# The digests are the TOP-LEVEL object. successResponse is the standard
# emitter but it does NOT add a wrapper — it is NextResponse.json(data)
# plus the no-store header — so there is no .data to reach through.
# A 200 that is not a manifest is how a hub which does not serve this path
# answers (the proxy rewrites unknown routes and returns HTML), so it
# means "wrong surface", not "corrupt".
if ! jq -e 'type == "object" and length > 0 and (to_entries | all(.value | type == "string"))' "$SCRIPT_MANIFEST" >/dev/null 2>&1; then
rm -f "$SCRIPT_MANIFEST"
return 1
fi
return 0
}
# load_script_manifest <group>
load_script_manifest() {
SCRIPT_GROUP="$1"
# "cmd; rc=$?" dies under the set -euo pipefail these steps run with —
# errexit fires before rc is read and the step ends with NO output. And
# "if ! cmd; then rc=$?" is worse: inside the branch $? is the status of
# the NEGATION (0), so every failure reads as success. "|| rc=$?" is the
# one form that both suppresses errexit and preserves the real code.
local rc=0
_try_manifest "$CI_SCRIPTS_URL" || rc=$?
if [ "$rc" = "0" ]; then
SCRIPTS_BASE_URL="$CI_SCRIPTS_URL"
echo "✅ script manifest loaded ($(jq -r 'length' "$SCRIPT_MANIFEST") scripts)"
return 0
fi
if [ "$rc" = "2" ]; then exit 1; fi
# No manifest on the scripts surface: a hub older than the manifest
# itself. The manifest is the file list, so there is nothing to download.
echo "❌ no script manifest on this hub ($HUB_BASE_URL): it cannot name the $SCRIPT_GROUP scripts. Redeploy the hub."
exit 1
}
# download_script_group <group> — the hub names the files, this workflow
# names only the group. Downloads every script of the group, in served order.
download_script_group() {
load_script_manifest "$1"
local name
# The loop runs in THIS shell (no pipe), so a failed download exits the step.
while IFS= read -r name; do
download_and_verify "$name"
done < <(jq -r 'keys_unsorted[]' "$SCRIPT_MANIFEST")
}
download_and_verify() {
local script_name="$1"
local output_path="/tmp/$script_name"
local expected_hash
expected_hash=$(jq -r --arg n "$script_name" '.[$n] // empty' "$SCRIPT_MANIFEST")
if [ -z "$expected_hash" ]; then
echo "❌ $script_name is not in the server's script manifest!"
echo " The hub serves no such script, or it failed to read on the server."
exit 1
fi
if ! printf '%s' "$expected_hash" | grep -Eq '^[0-9a-f]{64}$'; then
echo "❌ the manifest entry for $script_name is not a SHA-256 digest — refusing to run it."
exit 1
fi
curl -fsSL "$SCRIPTS_BASE_URL/$script_name" \
-K "$CURL_CFG" \
-o "$output_path"
local actual_hash=$(shasum -a 256 "$output_path" | cut -d' ' -f1)
if [ "$actual_hash" != "$expected_hash" ]; then
echo "❌ HASH MISMATCH for $script_name!"
echo " Expected: $expected_hash"
echo " Actual: $actual_hash"
echo " The download was corrupted in transit — both values come from the same deployment."
exit 1
fi
# Make shell scripts executable
if [[ "$script_name" == *.sh ]]; then
chmod +x "$output_path"
fi
echo "✅ $script_name verified (hash: ${actual_hash:0:16}...)"
}
# Digests AND the file list come from the deployment serving the bytes,
# not from this file: the step names a group (SCRIPT_GROUPS in the hub's
# lib/config/ci-script-catalog.ts) and downloads what the hub lists for it.
download_script_group "code-graph"
# LIVE change sets (a `pull_request` event): ask the hub to refresh NOW and
# whether this run builds the PR's overlay (`build`). Never fails the run.
# A fork has no secrets and is excluded by the job's `if:`, as are the hub
# tools' own branches (TOOL_BRANCH_PREFIXES); a bot's own description edit
# (the hub writing its block) is excluded too, or every hub write would
# start another run.
- name: Refresh the change set
id: refresh
if: github.event_name == 'pull_request' && steps.scripts.outcome == 'success'
# A hub that cannot answer never fails the pull request's check: the schedule links the set.
continue-on-error: true
env:
WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }}
GITHUB_REPOSITORY: ${{ github.repository }}
CODE_GRAPH_REFRESH_PR: ${{ github.event.pull_request.number }}
# The head of THIS event: the hub answers `build` only while it is still the PR's live head.
CODE_GRAPH_REFRESH_HEAD_SHA: ${{ github.event.pull_request.head.sha }}
# The event's action: the hub runs its job only for one that can change a set (CHANGE_SET_REFRESH_ACTIONS).
CODE_GRAPH_REFRESH_ACTION: ${{ github.event.action }}
# The event's sender type: the hub refuses a bot's own description edit too, not only this job's `if:`.
CODE_GRAPH_REFRESH_SENDER_TYPE: ${{ github.event.sender.type }}
run: node /tmp/code-graph-build.mjs
# FULL history, blobless. `collectFileFacts` derives per-file ownership
# (last commit, recent commits, commits in the window) from one
# `git log --no-merges --no-renames` walk, which a depth-1 checkout
# cannot answer — it would report every file as owned by one commit.
# `filter: blob:none` keeps the clone cheap: the walk reads commit
# metadata and name-only paths, never file contents, which is also why
# the walk passes `--no-renames` (rename detection would fetch blobs).
# A shallow checkout still degrades gracefully: ownership is omitted and
# `coverage.ownership` is false rather than the job failing.
# In overlay mode the checkout is the PR HEAD, with every blob: the
# overlay diffs it against the live snapshot commit with rename
# detection, which reads contents a blob:none clone cannot fetch here.
#
# The four build steps below share one `if:` (a PR event builds only when the refresh answered
# `build`) and one `continue-on-error` (a pull request's check never goes red on the overlay: the
# graph lane and the schedule rebuild it). Actions has no step group, so each step states both.
- name: Check out repository
if: github.event_name != 'pull_request' || steps.refresh.outputs.build == 'true'
continue-on-error: ${{ github.event_name == 'pull_request' }}
uses: actions/checkout@fbc6f3992d24b796d5a048ff273f7fcc4a7b6c09 # v5.1.0
with:
ref: ${{ env.OVERLAY_HEAD_SHA }}
fetch-depth: 0
# `!x && 'blob:none' || ''` — never `x && '' || …`: '' is falsy in an expression, so that form always yields blob:none.
filter: ${{ !env.OVERLAY_PR && 'blob:none' || '' }}
persist-credentials: false
- name: Set up Node.js
if: github.event_name != 'pull_request' || steps.refresh.outputs.build == 'true'
continue-on-error: ${{ github.event_name == 'pull_request' }}
uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
with:
node-version: '22'
# v5+ caches automatically when it finds a package manager; this job never did.
package-manager-cache: false
# The command is CODE_GRAPH_INSTALL_COMMAND (lib/config/code-graph-workflow.ts),
# the one spelling both workflows use: pinned wasm tree-sitter + grammars +
# yaml into an isolated tree under RUNNER_TEMP, exported as CODE_GRAPH_DEPS_DIR.
- name: Install graph dependencies
if: github.event_name != 'pull_request' || steps.refresh.outputs.build == 'true'
continue-on-error: ${{ github.event_name == 'pull_request' }}
run: mkdir -p "$RUNNER_TEMP/code-graph-deps" && cd "$RUNNER_TEMP/code-graph-deps" && printf '%s' '{"name":"code-graph-deps","version":"1.0.0","private":true,"dependencies":{"web-tree-sitter":"0.27.0","@vscode/tree-sitter-wasm":"0.3.1","yaml":"2.9.1"}}' > package.json && printf '%s' '{"name":"code-graph-deps","version":"1.0.0","lockfileVersion":3,"requires":true,"packages":{"":{"name":"code-graph-deps","version":"1.0.0","dependencies":{"web-tree-sitter":"0.27.0","@vscode/tree-sitter-wasm":"0.3.1","yaml":"2.9.1"}},"node_modules/web-tree-sitter":{"version":"0.27.0","resolved":"https://registry.npmjs.org/web-tree-sitter/-/web-tree-sitter-0.27.0.tgz","integrity":"sha512-XK08gj6RwTMQatAG7uVRP8MunqotL/XC19vHgkSPKmELgbGPBj4ECvB8haHOUnyj6ls2B8t42UTro14zxGgAHg=="},"node_modules/@vscode/tree-sitter-wasm":{"version":"0.3.1","resolved":"https://registry.npmjs.org/@vscode/tree-sitter-wasm/-/tree-sitter-wasm-0.3.1.tgz","integrity":"sha512-RJFoomET6FajjG511fmQxeBQfU6M24a0aFZPqpid+ttIxanWf1VGytBG0UmsGjt07qmIPJS8U31D+aecuCucsQ=="},"node_modules/yaml":{"version":"2.9.1","resolved":"https://registry.npmjs.org/yaml/-/yaml-2.9.1.tgz","integrity":"sha512-3NxN8+78OdzbT7C/WjGsyfPAtJaN3FNDsWxv7Y7mcDsT/oOmgW8BpyQQFFBnvZE3j9Y2Sdz1ULFLezL7Eb2yFw=="}}}' > package-lock.json && npm ci --ignore-scripts --no-audit --no-fund && echo "CODE_GRAPH_DEPS_DIR=$RUNNER_TEMP/code-graph-deps" >> "$GITHUB_ENV" || { echo "::warning::graph dependencies failed their lockfile-enforced install; continuing without them"; rm -rf "$RUNNER_TEMP/code-graph-deps"; exit 1; }
# Overlay mode posts the PR's overlay INSTEAD of a snapshot (code-graph-build.mjs `overlayMain`).
- name: Build and upload the code graph
id: graph
if: github.event_name != 'pull_request' || steps.refresh.outputs.build == 'true'
continue-on-error: ${{ github.event_name == 'pull_request' }}
env:
WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }}
CODE_GRAPH_DEPS_DIR: ${{ env.CODE_GRAPH_DEPS_DIR }}
GITHUB_REPOSITORY: ${{ github.repository }}
# Overlay mode: the job's overlay coordinates (empty outside it).
CODE_GRAPH_OVERLAY_PR: ${{ env.OVERLAY_PR }}
CODE_GRAPH_OVERLAY_HEAD_SHA: ${{ env.OVERLAY_HEAD_SHA }}
# An overlay run names a NON-default branch, so a hub or builder that predates overlay mode lands
# the PR head as a `branch` snapshot, never as the repository's live graph (the e2e run found an
# older builder promoting an unmerged PR's head live). Empty outside overlay mode.
CODE_GRAPH_BRANCH: ${{ env.OVERLAY_PR && format('overlay/pr-{0}', env.OVERLAY_PR) || '' }}
run: node /tmp/code-graph-build.mjs
# ===========================================================================
# DOCUMENTATION PIPELINE: stages 0 to 4, one pull request per run. Every
# stage commits and pushes its own output as it finishes, so a timeout or a
# cancellation loses at most the stage in flight.
# ===========================================================================
doc-pipeline:
runs-on: ubuntu-latest
timeout-minutes: 720 # 12 hours for large repositories with many files
# contents: push the docs branch. pull-requests: open and update the pull
# request. issues: `gh label create` for the documentation / automated /
# in-progress labels; without it a repository that lacks them answers 403
# on the label and then 422 on `gh pr create --label`.
permissions:
contents: write
pull-requests: write
issues: write
# Never on push (a push registers the workflow and runs the code-graph job
# only), never on the graph-only re-dispatch, never on a graph-only manual
# run. This is what lets workflow_dispatch API calls work on feature branches.
if: github.event_name != 'push' && github.event_name != 'pull_request' && github.event.action != 'flamingo-code-graph' && github.event.inputs.graph_only != 'true'
steps:
# =========================================================================
# REPORT CAPABILITY FIRST (shared failure-net standard with the code-review
# workflow): workflow-helpers.sh — which carries send_webhook and the
# report/stage-callback helpers — downloads in its OWN step before anything
# else, so a failure in the main script download below can still be pinged
# and reported home instead of leaving a phantom pending/running row.
# =========================================================================
- name: Download report helpers
env:
WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }}
run: |
# Function to download and verify script
SCRIPT_MANIFEST=/tmp/flamingo-script-manifest.json
# WEBHOOK_SECRET reaches curl through a 0600 config file, never argv — see
# curlAuthPreamble, which always traps the removal.
CURL_CFG=$(mktemp) && chmod 600 "$CURL_CFG"
trap 'rm -f "$CURL_CFG"' EXIT
printf 'header = "Authorization: Bearer %s"\n' "$WEBHOOK_SECRET" > "$CURL_CFG"
# The scripts surface. load_script_manifest pins SCRIPTS_BASE_URL to it.
CI_SCRIPTS_URL="${HUB_BASE_URL%/}/api/ci/scripts"
# _try_manifest <base> — 0 loaded, 1 no manifest surface there, 2 fatal.
# The manifest is asked for ONE group: its keys are the files to download.
_try_manifest() {
local base="$1" code
code=$(curl -sS -w '%{http_code}' -o "$SCRIPT_MANIFEST" \
-K "$CURL_CFG" \
"$base/manifest.json?group=$SCRIPT_GROUP") || code="000"
if [ "$code" = "404" ]; then rm -f "$SCRIPT_MANIFEST"; return 1; fi
if [ "$code" != "200" ]; then
echo "❌ manifest request to $base failed (HTTP $code)"
rm -f "$SCRIPT_MANIFEST"
return 2
fi
# The digests are the TOP-LEVEL object. successResponse is the standard
# emitter but it does NOT add a wrapper — it is NextResponse.json(data)
# plus the no-store header — so there is no .data to reach through.
# A 200 that is not a manifest is how a hub which does not serve this path
# answers (the proxy rewrites unknown routes and returns HTML), so it
# means "wrong surface", not "corrupt".
if ! jq -e 'type == "object" and length > 0 and (to_entries | all(.value | type == "string"))' "$SCRIPT_MANIFEST" >/dev/null 2>&1; then
rm -f "$SCRIPT_MANIFEST"
return 1
fi
return 0
}
# load_script_manifest <group>
load_script_manifest() {
SCRIPT_GROUP="$1"
# "cmd; rc=$?" dies under the set -euo pipefail these steps run with —
# errexit fires before rc is read and the step ends with NO output. And
# "if ! cmd; then rc=$?" is worse: inside the branch $? is the status of
# the NEGATION (0), so every failure reads as success. "|| rc=$?" is the
# one form that both suppresses errexit and preserves the real code.
local rc=0
_try_manifest "$CI_SCRIPTS_URL" || rc=$?
if [ "$rc" = "0" ]; then
SCRIPTS_BASE_URL="$CI_SCRIPTS_URL"
echo "✅ script manifest loaded ($(jq -r 'length' "$SCRIPT_MANIFEST") scripts)"
return 0
fi
if [ "$rc" = "2" ]; then exit 1; fi
# No manifest on the scripts surface: a hub older than the manifest
# itself. The manifest is the file list, so there is nothing to download.
echo "❌ no script manifest on this hub ($HUB_BASE_URL): it cannot name the $SCRIPT_GROUP scripts. Redeploy the hub."
exit 1
}
# download_script_group <group> — the hub names the files, this workflow
# names only the group. Downloads every script of the group, in served order.
download_script_group() {
load_script_manifest "$1"
local name
# The loop runs in THIS shell (no pipe), so a failed download exits the step.
while IFS= read -r name; do
download_and_verify "$name"
done < <(jq -r 'keys_unsorted[]' "$SCRIPT_MANIFEST")
}
download_and_verify() {
local script_name="$1"
local output_path="/tmp/$script_name"
local expected_hash
expected_hash=$(jq -r --arg n "$script_name" '.[$n] // empty' "$SCRIPT_MANIFEST")
if [ -z "$expected_hash" ]; then
echo "❌ $script_name is not in the server's script manifest!"
echo " The hub serves no such script, or it failed to read on the server."
exit 1
fi
if ! printf '%s' "$expected_hash" | grep -Eq '^[0-9a-f]{64}$'; then
echo "❌ the manifest entry for $script_name is not a SHA-256 digest — refusing to run it."
exit 1
fi
curl -fsSL "$SCRIPTS_BASE_URL/$script_name" \
-K "$CURL_CFG" \
-o "$output_path"
local actual_hash=$(shasum -a 256 "$output_path" | cut -d' ' -f1)
if [ "$actual_hash" != "$expected_hash" ]; then
echo "❌ HASH MISMATCH for $script_name!"
echo " Expected: $expected_hash"
echo " Actual: $actual_hash"
echo " The download was corrupted in transit — both values come from the same deployment."
exit 1
fi
# Make shell scripts executable
if [[ "$script_name" == *.sh ]]; then
chmod +x "$output_path"
fi
echo "✅ $script_name verified (hash: ${actual_hash:0:16}...)"
}
# Digests AND the file list come from the deployment serving the bytes, not from this file.
download_script_group "doc-helpers"
# =========================================================================
# REPORT RUN STARTED: the early "the workflow actually started" ping.
# Deliberately BEFORE the main script download: it stamps workflow_run_id +
# status 'running' on the hub's run row, which is what lets the hub's tiered
# reaper tell "dispatch accepted but nothing ran" (never pinged, failed
# fast) from "started and then crashed" (pinged, longer deadline).
# =========================================================================
- name: Report run started
if: env.HUB_BASE_URL != ''
continue-on-error: true
env:
WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }}
run: |
source /tmp/workflow-helpers.sh
CALLBACK_URL="${HUB_BASE_URL}/api/code-documentation/webhook"
echo "Reporting run start to $CALLBACK_URL"
PAYLOAD="{
\"run_id\": \"$RUN_ID\",
\"repo_id\": \"$REPO_ID\",
\"status\": \"running\",
\"workflow_run_id\": ${{ github.run_id }},
\"workflow_url\": \"${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}\",
\"current_stage\": \"inline-docs\"
}"
HTTP_CODE=$(send_webhook "$CALLBACK_URL" "$WEBHOOK_SECRET" "$PAYLOAD" "/tmp/webhook_start_response.txt") || HTTP_CODE="failed"
if [ "$HTTP_CODE" = "200" ] || [ "$HTTP_CODE" = "201" ]; then
echo "Hub acknowledged the run start (HTTP $HTTP_CODE)"
else
echo "::warning title=Run start not reported::The hub answered HTTP $HTTP_CODE to the start callback. The run continues; the hub learns its status from the stage callbacks."
fi
# =========================================================================
# DOWNLOAD PIPELINE SCRIPTS
# Every script a documentation run uses, from the hub's authenticated
# /api/ci/scripts surface (workflow-helpers.sh arrived in "Download report
# helpers"), plus the source vocabulary and the Markdown guidelines.
# =========================================================================
- name: Download pipeline scripts
id: helpers
env:
WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }}
# No HASH_* pins: the digests come from manifest.json on the same
# endpoint that serves the scripts, so a hash in this file can never
# be a different ref's than the bytes it checks.
run: |
echo "::group::Download the doc-pipeline script group"
# Function to download and verify script
SCRIPT_MANIFEST=/tmp/flamingo-script-manifest.json
# WEBHOOK_SECRET reaches curl through a 0600 config file, never argv — see
# curlAuthPreamble, which always traps the removal.
CURL_CFG=$(mktemp) && chmod 600 "$CURL_CFG"
trap 'rm -f "$CURL_CFG"' EXIT
printf 'header = "Authorization: Bearer %s"\n' "$WEBHOOK_SECRET" > "$CURL_CFG"
# The scripts surface. load_script_manifest pins SCRIPTS_BASE_URL to it.
CI_SCRIPTS_URL="${HUB_BASE_URL%/}/api/ci/scripts"
# _try_manifest <base> — 0 loaded, 1 no manifest surface there, 2 fatal.
# The manifest is asked for ONE group: its keys are the files to download.
_try_manifest() {
local base="$1" code
code=$(curl -sS -w '%{http_code}' -o "$SCRIPT_MANIFEST" \
-K "$CURL_CFG" \
"$base/manifest.json?group=$SCRIPT_GROUP") || code="000"
if [ "$code" = "404" ]; then rm -f "$SCRIPT_MANIFEST"; return 1; fi
if [ "$code" != "200" ]; then
echo "❌ manifest request to $base failed (HTTP $code)"
rm -f "$SCRIPT_MANIFEST"
return 2
fi
# The digests are the TOP-LEVEL object. successResponse is the standard
# emitter but it does NOT add a wrapper — it is NextResponse.json(data)
# plus the no-store header — so there is no .data to reach through.
# A 200 that is not a manifest is how a hub which does not serve this path
# answers (the proxy rewrites unknown routes and returns HTML), so it
# means "wrong surface", not "corrupt".
if ! jq -e 'type == "object" and length > 0 and (to_entries | all(.value | type == "string"))' "$SCRIPT_MANIFEST" >/dev/null 2>&1; then
rm -f "$SCRIPT_MANIFEST"
return 1
fi
return 0
}
# load_script_manifest <group>
load_script_manifest() {
SCRIPT_GROUP="$1"
# "cmd; rc=$?" dies under the set -euo pipefail these steps run with —
# errexit fires before rc is read and the step ends with NO output. And
# "if ! cmd; then rc=$?" is worse: inside the branch $? is the status of
# the NEGATION (0), so every failure reads as success. "|| rc=$?" is the
# one form that both suppresses errexit and preserves the real code.
local rc=0
_try_manifest "$CI_SCRIPTS_URL" || rc=$?
if [ "$rc" = "0" ]; then
SCRIPTS_BASE_URL="$CI_SCRIPTS_URL"
echo "✅ script manifest loaded ($(jq -r 'length' "$SCRIPT_MANIFEST") scripts)"
return 0
fi
if [ "$rc" = "2" ]; then exit 1; fi
# No manifest on the scripts surface: a hub older than the manifest
# itself. The manifest is the file list, so there is nothing to download.
echo "❌ no script manifest on this hub ($HUB_BASE_URL): it cannot name the $SCRIPT_GROUP scripts. Redeploy the hub."
exit 1
}
# download_script_group <group> — the hub names the files, this workflow
# names only the group. Downloads every script of the group, in served order.
download_script_group() {
load_script_manifest "$1"
local name
# The loop runs in THIS shell (no pipe), so a failed download exits the step.
while IFS= read -r name; do
download_and_verify "$name"
done < <(jq -r 'keys_unsorted[]' "$SCRIPT_MANIFEST")
}
download_and_verify() {
local script_name="$1"
local output_path="/tmp/$script_name"
local expected_hash
expected_hash=$(jq -r --arg n "$script_name" '.[$n] // empty' "$SCRIPT_MANIFEST")
if [ -z "$expected_hash" ]; then
echo "❌ $script_name is not in the server's script manifest!"
echo " The hub serves no such script, or it failed to read on the server."
exit 1
fi
if ! printf '%s' "$expected_hash" | grep -Eq '^[0-9a-f]{64}$'; then
echo "❌ the manifest entry for $script_name is not a SHA-256 digest — refusing to run it."
exit 1
fi
curl -fsSL "$SCRIPTS_BASE_URL/$script_name" \
-K "$CURL_CFG" \
-o "$output_path"
local actual_hash=$(shasum -a 256 "$output_path" | cut -d' ' -f1)
if [ "$actual_hash" != "$expected_hash" ]; then
echo "❌ HASH MISMATCH for $script_name!"
echo " Expected: $expected_hash"
echo " Actual: $actual_hash"
echo " The download was corrupted in transit — both values come from the same deployment."
exit 1
fi
# Make shell scripts executable
if [[ "$script_name" == *.sh ]]; then
chmod +x "$output_path"
fi
echo "✅ $script_name verified (hash: ${actual_hash:0:16}...)"
}
# Every script a documentation run uses, stage 0 (the code graph) included.
# Digests AND the file list come from the deployment serving the bytes,
# not from this file: the hub lists the group in download order (a helper
# a generator require()s at load comes before it).
download_script_group "doc-pipeline"
echo "::endgroup::"
# ONE vocabulary read for the whole run: every later step (language
# detection, source discovery, the generators, the graph build) reads this
# file, so none of them needs the secret for it.
node /tmp/ci-source.mjs vocabulary /tmp/ci-vocabulary.json || { echo "::error title=Source vocabulary unavailable::Could not read the source vocabulary from the hub."; exit 1; }
echo "CODE_GRAPH_VOCABULARY_FILE=/tmp/ci-vocabulary.json" >> $GITHUB_ENV
echo "Pipeline scripts and source vocabulary downloaded and verified"
# Export paths for all stages (use os.tmpdir() compatible paths)
echo "VALIDATION_RULES_PATH=/tmp/markdown-validation-rules.md" >> $GITHUB_ENV
echo "GUIDELINES_PATH=/tmp/flamingo-markdown-guidelines.md" >> $GITHUB_ENV
echo "STAGE3_FILES_TRACKER=/tmp/stage3-files.txt" >> $GITHUB_ENV
echo "STAGE3_STATS_FILE=/tmp/.doc-stage3-stats.json" >> $GITHUB_ENV
echo "STAGE4_FILES_TRACKER=/tmp/stage4-files.txt" >> $GITHUB_ENV
# Flamingo Markdown guidelines (their own endpoint). REQUIRED: the
# Markdown validation and the CodeWiki prompts both read them.
GUIDELINES_URL="${HUB_BASE_URL}/api/code-documentation/guidelines"
# Same 0600 config file the download block above set up.
HTTP_CODE=$(curl -fsSL -w "%{http_code}" \
"$GUIDELINES_URL" \
-K "$CURL_CFG" \
-o "/tmp/flamingo-markdown-guidelines.md" 2>/dev/null) || HTTP_CODE="failed"
if [ "$HTTP_CODE" = "200" ]; then
GUIDELINES_SIZE=$(wc -c < /tmp/flamingo-markdown-guidelines.md | tr -d ' ')
if [ "$GUIDELINES_SIZE" -lt 100 ]; then
echo "::error title=Markdown guidelines invalid::The guidelines file is $GUIDELINES_SIZE bytes, which is an error response, not guidelines. Its body follows."
cat /tmp/flamingo-markdown-guidelines.md
exit 1
fi
echo "Markdown guidelines downloaded ($GUIDELINES_SIZE bytes)"
else
echo "::error title=Markdown guidelines unavailable::GET $GUIDELINES_URL answered HTTP $HTTP_CODE. The Markdown validation and the CodeWiki prompts require them; check that the hub serves the guidelines endpoint."
rm -f /tmp/flamingo-markdown-guidelines.md
exit 1
fi
# (The run-started report sits ABOVE the main script download; see the
# report-capability step ordering at the top of the job.)
- name: Check out repository
# v5 = the Node 24 drop-in (v4 targets EOL Node 20 and warns on every run).
uses: actions/checkout@fbc6f3992d24b796d5a048ff273f7fcc4a7b6c09 # v5.1.0
with:
fetch-depth: 0
token: ${{ secrets.GITHUB_TOKEN }}
ref: ${{ env.SOURCE_BRANCH }}
# The SOURCE head, before the PR branch and the docs-removal commit move
# HEAD: the stage-0 graph build tags this commit (the code being
# documented), never the docs branch it is sitting on.
- name: Record source head
run: echo "SOURCE_HEAD_SHA=$(git rev-parse HEAD)" >> "$GITHUB_ENV"
# =========================================================================
# DEPENDENCY REPOSITORIES (when configured): cloned beside the checkout as
# context for every stage.
# =========================================================================
- name: Clone dependency repositories
if: env.DEPENDENCIES != ''
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
# Authenticates the clone-token request to the hub (masked by the runner).
WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }}
run: |
source /tmp/workflow-helpers.sh
echo "Cloning dependency repositories: $DEPENDENCIES"
ensure_directory "../deps"
# The clone credential is minted by the hub for THIS run: read-only,
# scoped to exactly this repository's dependencies, valid one hour.
# A stored repository secret cannot hold it (an installation token
# expires an hour after it is written).
CLONE_TOKEN=""
TOKEN_FILE=$(mktemp) && chmod 600 "$TOKEN_FILE"
HTTP_CODE=$(node /tmp/ci-hub.mjs get "/api/ci/code-documentation/clone-token?github=${GITHUB_REPOSITORY}" "$TOKEN_FILE") || HTTP_CODE="000"
if [ "$HTTP_CODE" = "200" ]; then
CLONE_TOKEN=$(jq -r '.token // empty' "$TOKEN_FILE")
fi
rm -f "$TOKEN_FILE"
if [ -n "$CLONE_TOKEN" ]; then
echo "::add-mask::$CLONE_TOKEN"
echo "Credential: a read-only token the hub minted for this run's dependencies"
else
echo "::warning title=No clone token::The hub did not mint a clone token (HTTP $HTTP_CODE); using GITHUB_TOKEN, which reads public repositories only."
CLONE_TOKEN="$GH_TOKEN"
fi
# The credential applies to THESE clones only (`git -c`), never globally:
# the token reads the dependencies and nothing else, so a global URL
# rewrite would also send this job's own pushes through it.
IFS=',' read -ra DEPS <<< "$DEPENDENCIES"
for dep in "${DEPS[@]}"; do
repo_name=$(basename $dep)
echo "::group::Clone $dep into ../deps/$repo_name"
if git -c "url.https://x-access-token:${CLONE_TOKEN}@github.com/.insteadOf=https://github.com/" \
clone --depth 1 "https://github.com/$dep.git" "../deps/$repo_name" 2>&1; then
git -C "../deps/$repo_name" remote set-url origin "https://github.com/$dep.git"
file_count=$(node /tmp/ci-source.mjs count "../deps/$repo_name" 2>/dev/null || echo "?")
echo "Cloned $dep ($file_count source files)"
else
echo "::warning title=Dependency not cloned::Could not clone $dep."
fi
echo "::endgroup::"
done
echo "::group::Dependency directories"
ls -la ../deps/ 2>/dev/null || echo "No dependencies cloned"
echo "::endgroup::"
echo "Dependency source files available as context: $(node /tmp/ci-source.mjs count ../deps 2>/dev/null || echo '?')"
# =========================================================================
# PRIMARY LANGUAGE (before every stage, so all of them agree): picks the
# stage 2 engine and filters stages 1 to 3.
# =========================================================================
- name: Detect primary language
id: detect_language
run: |
source /tmp/workflow-helpers.sh
# This step is the FIRST reader of CODEWIKI_CONFIG_JSON — it runs before
# "Validate and parse run configuration", so the emptiness guard lives here,
# ahead of the first jq, rather than in the later validation step.
if [ -z "$CODEWIKI_CONFIG_JSON" ]; then
echo "::error title=Missing configuration::CODEWIKI_CONFIG_JSON is empty; language detection needs it."
exit 1
fi
echo "Detecting the primary language across the repository (.) and its dependencies (../deps/)"
# ONE detection, from the vocabulary the hub serves (ci-source.mjs): which
# languages are source, their extensions, and which of them CodeWiki can
# parse are rows in the hub's language table. This step used to carry nine
# hand-typed `find` counts, a positional helper and a six-way threshold
# test, each with its own idea of the extensions and the exclusions.
DETECTION=$(node /tmp/ci-source.mjs detect) || { echo "::error title=Language detection failed::ci-source.mjs detect exited non-zero."; exit 1; }
PRIMARY_LANG=$(echo "$DETECTION" | jq -r '.primary')
MAX_COUNT=$(echo "$DETECTION" | jq -r '.max')
CODEWIKI_SUPPORTED=$(echo "$DETECTION" | jq -r '.codewiki_supported')
echo "::group::Source files by language (tests and never-source directories excluded)"
echo "$DETECTION" | jq -r '.counts | to_entries[] | select(.value > 0) | " \(.key): \(.value)"'
echo "::endgroup::"
echo "Primary language: $PRIMARY_LANG ($MAX_COUNT files)"
if [ "$CODEWIKI_SUPPORTED" = "true" ]; then
echo "CodeWiki can parse it: yes"
else
echo "CodeWiki can parse it: no (stage 2 uses the Claude architecture analysis)"
fi
# Per-repo engine override.
#
# The detection above cannot see mixed repos: it counts `.` AND `../deps`, so
# a Rust or Go product with a TypeScript dependency clones its way past the
# >=10 threshold and runs CodeWiki over a codebase whose analyzers do not
# exist — which yields synthetic module_1/module_2/... docs that look like a
# successful run. `engine` pins the choice.
#
# Applied here, before set_output, so all five downstream gates keep reading
# one value and need no change. It cannot live in the `if:` conditions:
# GitHub Actions expressions have no ternary.
CODEWIKI_ENGINE=$(require_json_key "$CODEWIKI_CONFIG_JSON" '.engine' 'codewiki engine') || exit 1
case "$CODEWIKI_ENGINE" in
claude)
CODEWIKI_SUPPORTED="false"
echo "Stage 2 engine: claude (pinned by the repository configuration)"
;;
codewiki)
CODEWIKI_SUPPORTED="true"
echo "Stage 2 engine: codewiki (pinned by the repository configuration)"
;;
auto)
echo "Stage 2 engine: auto (CodeWiki supported: $CODEWIKI_SUPPORTED)"
;;
*)
echo "::error title=Invalid stage 2 engine::engine is '$CODEWIKI_ENGINE'; expected auto, codewiki or claude."
exit 1
;;
esac
# Output for use by subsequent steps
set_output "primary_language" "$PRIMARY_LANG"
set_output "codewiki_supported" "$CODEWIKI_SUPPORTED"
set_output "file_count" "$MAX_COUNT"
# =========================================================================
# VALIDATE AND PARSE THE RUN CONFIGURATION
# 1. Every required parameter is present
# 2. Each JSON configuration is parsed into individual variables (the JSON
# grouping keeps the dispatch under GitHub's 25-input limit)
# 3. The parsed values are valid, then exported to GITHUB_ENV
# =========================================================================
- name: Validate and parse run configuration
run: |
# 1. Required parameters -------------------------------------------
VALIDATION_FAILED=0
# Core parameters
[ -z "$RUN_ID" ] && echo "::error title=Missing parameter::RUN_ID" && VALIDATION_FAILED=1
[ -z "$REPO_ID" ] && echo "::error title=Missing parameter::REPO_ID" && VALIDATION_FAILED=1
[ -z "$HUB_BASE_URL" ] && echo "::error title=Missing parameter::HUB_BASE_URL" && VALIDATION_FAILED=1
[ -z "$STAGES" ] && echo "::error title=Missing parameter::STAGES" && VALIDATION_FAILED=1
[ -z "$CLAUDE_MODEL" ] && echo "::error title=Missing parameter::CLAUDE_MODEL" && VALIDATION_FAILED=1
# JSON parameters
[ -z "$CODEWIKI_CONFIG_JSON" ] && echo "::error title=Missing parameter::CODEWIKI_CONFIG_JSON" && VALIDATION_FAILED=1
[ -z "$OUTPUT_PATHS_JSON" ] && echo "::error title=Missing parameter::OUTPUT_PATHS_JSON" && VALIDATION_FAILED=1
[ -z "$README_CONFIG_JSON" ] && echo "::error title=Missing parameter::README_CONFIG_JSON" && VALIDATION_FAILED=1
[ -z "$YOUTUBE_CONFIG_JSON" ] && echo "::error title=Missing parameter::YOUTUBE_CONFIG_JSON" && VALIDATION_FAILED=1
if [ $VALIDATION_FAILED -eq 1 ]; then
echo "::error title=Invalid run configuration::Required parameters are missing (annotated above). The hub sends every one of them; re-run \"Setup Workflow\" if this repository's workflow is out of date."
exit 1
fi
echo "Required parameters: present"
# 2a. CodeWiki configuration (nested JSON) ---------------------------
echo "::group::CodeWiki configuration"
# Only the keys with a real consumer are extracted here — the per-phase
# base_url / api_version / temperature / temperature_supported are read
# directly from CODEWIKI_CONFIG_JSON by configure_codewiki_from_json, which
# is the single place that builds the `codewiki config set` command. They
# used to be parsed here as well and exported to $GITHUB_ENV, where nothing
# read them.
#
# No `// default` fallbacks anywhere below. The hub deep-merges every JSON
# param against the params SSOT before dispatch, so an absent key is a real
# bug — and a fallback here would silently win over the SSOT, which is how
# docs/architecture and docs/reference/architecture drifted apart.
# `require_json_key` / `optional_json_key` come from workflow-helpers.sh.
source /tmp/workflow-helpers.sh
CW_JSON="$CODEWIKI_CONFIG_JSON"
# Parse nested cluster config
CODEWIKI_CLUSTER_PROVIDER=$(require_json_key "$CW_JSON" '.cluster.provider' 'cluster provider') || exit 1
CODEWIKI_CLUSTER_MODEL=$(require_json_key "$CW_JSON" '.cluster.model' 'cluster model') || exit 1
CODEWIKI_CLUSTER_MAX_TOKENS=$(require_json_key "$CW_JSON" '.cluster.max_tokens' 'cluster max_tokens') || exit 1
CODEWIKI_CLUSTER_MAX_TOKEN_FIELD=$(require_json_key "$CW_JSON" '.cluster.max_token_field' 'cluster max_token_field') || exit 1
# Nullable by design: api_version is null for every OpenAI model.
# Parse nested generation config
CODEWIKI_GENERATION_PROVIDER=$(require_json_key "$CW_JSON" '.generation.provider' 'generation provider') || exit 1
CODEWIKI_GENERATION_MODEL=$(require_json_key "$CW_JSON" '.generation.model' 'generation model') || exit 1
CODEWIKI_GENERATION_MAX_TOKENS=$(require_json_key "$CW_JSON" '.generation.max_tokens' 'generation max_tokens') || exit 1
CODEWIKI_GENERATION_MAX_TOKEN_FIELD=$(require_json_key "$CW_JSON" '.generation.max_token_field' 'generation max_token_field') || exit 1
# Parse nested fallback config
CODEWIKI_FALLBACK_PROVIDER=$(require_json_key "$CW_JSON" '.fallback.provider' 'fallback provider') || exit 1
CODEWIKI_FALLBACK_MODEL=$(require_json_key "$CW_JSON" '.fallback.model' 'fallback model') || exit 1
CODEWIKI_FALLBACK_MAX_TOKENS=$(require_json_key "$CW_JSON" '.fallback.max_tokens' 'fallback max_tokens') || exit 1
CODEWIKI_FALLBACK_MAX_TOKEN_FIELD=$(require_json_key "$CW_JSON" '.fallback.max_token_field' 'fallback max_token_field') || exit 1
# Parse top-level config.
# max_files_per_module is LIVE: this value reaches CodeWiki through the
# job-scoped $GITHUB_ENV write below, and upstream reads it in its
# empty-module-tree branch — the branch Go/Rust/HCL repos land in.
CODEWIKI_MAX_FILES_PER_MODULE=$(require_json_key "$CW_JSON" '.max_files_per_module' 'max_files_per_module') || exit 1
CODEWIKI_MAX_DEPTH=$(require_json_key "$CW_JSON" '.max_depth' 'max_depth') || exit 1
CODEWIKI_REPO=$(require_json_key "$CW_JSON" '.repo' 'codewiki repo url') || exit 1
echo "Cluster (phase 2): $CODEWIKI_CLUSTER_PROVIDER / $CODEWIKI_CLUSTER_MODEL (${CODEWIKI_CLUSTER_MAX_TOKEN_FIELD}, ${CODEWIKI_CLUSTER_MAX_TOKENS} tokens)"
echo "Generation (phase 3+): $CODEWIKI_GENERATION_PROVIDER / $CODEWIKI_GENERATION_MODEL (${CODEWIKI_GENERATION_MAX_TOKEN_FIELD}, ${CODEWIKI_GENERATION_MAX_TOKENS} tokens)"
echo "Fallback: $CODEWIKI_FALLBACK_PROVIDER / $CODEWIKI_FALLBACK_MODEL (${CODEWIKI_FALLBACK_MAX_TOKEN_FIELD}, ${CODEWIKI_FALLBACK_MAX_TOKENS} tokens)"
echo "Max depth: $CODEWIKI_MAX_DEPTH"
echo "Max files per module: $CODEWIKI_MAX_FILES_PER_MODULE"
echo "::endgroup::"
# 2b. YouTube configuration ------------------------------------------
echo "::group::YouTube configuration"
# Parse YouTube config from JSONB (channels only - API key from secrets)
# channels is legitimately optional: no channels == feature off
YOUTUBE_CHANNELS=$(echo "$YOUTUBE_CONFIG_JSON" | jq -c '.channels // []')
# YouTube is enabled if channels array has items
YOUTUBE_ENABLED=$(echo "$YOUTUBE_CHANNELS" | jq -r 'if length > 0 then "true" else "false" end')
echo "Enabled: $YOUTUBE_ENABLED (from the channel count)"
echo "Channels: $YOUTUBE_CHANNELS"
echo "API key: secrets.YOUTUBE_API_KEY (never stored on the hub)"
echo "::endgroup::"
# 2c. README logo configuration --------------------------------------
echo "::group::README logo configuration"
# Parse README config from JSONB
README_LOGO_DARK=$(optional_json_key "$README_CONFIG_JSON" '.logo_dark')
README_LOGO_LIGHT=$(optional_json_key "$README_CONFIG_JSON" '.logo_light')
README_LOGO_ALT=$(require_json_key "$README_CONFIG_JSON" '.logo_alt' 'readme logo alt') || exit 1
echo "Dark logo: ${README_LOGO_DARK:-'(not set)'}"
echo "Light logo: ${README_LOGO_LIGHT:-'(not set)'}"
echo "Alt text: $README_LOGO_ALT"
echo "::endgroup::"
# 2d. Output paths ---------------------------------------------------
echo "::group::Output paths"
# Same fail-loud contract as the codewiki block. These fallbacks were the
# last surviving copy of the five paths, and `.reference` still said
# docs/architecture while the SSOT said docs/reference/architecture — the
# very drift the SSOT was created to end.
OP_JSON="$OUTPUT_PATHS_JSON"
DOCS_OUTPUT_PATH=$(require_json_key "$OP_JSON" '.docs' 'docs output path') || exit 1
REFERENCE_OUTPUT_PATH=$(require_json_key "$OP_JSON" '.reference' 'reference output path') || exit 1
DIAGRAMS_OUTPUT_PATH=$(require_json_key "$OP_JSON" '.diagrams' 'diagrams output path') || exit 1
GETTING_STARTED_OUTPUT_PATH=$(require_json_key "$OP_JSON" '.getting_started' 'getting-started output path') || exit 1
DEVELOPMENT_OUTPUT_PATH=$(require_json_key "$OP_JSON" '.development' 'development output path') || exit 1
echo "Docs root: $DOCS_OUTPUT_PATH"
echo "Reference: $REFERENCE_OUTPUT_PATH"
echo "Diagrams: $DIAGRAMS_OUTPUT_PATH"
echo "Getting started: $GETTING_STARTED_OUTPUT_PATH"
echo "Development: $DEVELOPMENT_OUTPUT_PATH"
echo "::endgroup::"
# 2e. Custom instructions and external repositories ------------------
echo "::group::Custom instructions and external repositories"
# Repository context (prevents invented GitHub URLs in every prompt)
# Extract repository information from GitHub context
GITHUB_REPO="${{ github.repository }}"
GITHUB_OWNER="${{ github.repository_owner }}"
GITHUB_SERVER="${{ github.server_url }}"
GITHUB_REPO_NAME=$(echo "$GITHUB_REPO" | cut -d'/' -f2)
GITHUB_REPO_URL="${GITHUB_SERVER}/${GITHUB_REPO}"
# Build repository context section (injected into ALL AI prompts)
# Use printf for multi-line string (avoids YAML parsing issues with heredoc)
printf -v REPOSITORY_CONTEXT '%s\n' \
'## REPOSITORY CONTEXT - GROUND TRUTH' \
'' \
'**CRITICAL:** This section provides the ACTUAL repository information. You MUST use these exact values when constructing GitHub URLs.' \
'' \
"- **Repository:** ${GITHUB_REPO}" \
"- **Owner:** ${GITHUB_OWNER}" \
"- **Repository Name:** ${GITHUB_REPO_NAME}" \
"- **Repository URL:** ${GITHUB_REPO_URL}" \
"- **Server:** ${GITHUB_SERVER}" \
'' \
'**MANDATORY RULES FOR GITHUB URLS:**' \
"1. ALWAYS use the exact repository path: \`${GITHUB_REPO}\`" \
'2. NEVER use placeholder URLs like "your-org", "example-org", or "mycompany"' \
'3. NEVER infer repository owner from file contents or dependencies' \
'4. NEVER use upstream/parent repository URLs (if this is a fork, use the fork URL)' \
"5. When linking to code: \`${GITHUB_REPO_URL}/blob/main/path/to/file\`" \
"6. When linking to clone: \`git clone ${GITHUB_REPO_URL}.git\`" \
"7. When linking to issues/PRs: \`${GITHUB_REPO_URL}/issues\` or \`${GITHUB_REPO_URL}/pulls\`" \
"8. When linking to releases: \`${GITHUB_REPO_URL}/releases\`" \
'' \
'**If you find yourself writing a GitHub URL, verify it matches the Repository URL above.**' \
"**ESPECIALLY IN README.md and tutorials - All GitHub URLs MUST use ${GITHUB_REPO}**" \
'' \
'---'
echo "Repository: $GITHUB_REPO"
echo "Repository URL: $GITHUB_REPO_URL"
# Repository context goes in front of the custom instructions
# Custom instructions come as plain text from user
# Read from the environment rather than interpolating into single quotes.
# GitHub Actions expression substitution runs BEFORE bash parses the line, so a
# single apostrophe anywhere in an admin's instructions used to terminate the
# string and kill the step with a syntax error. NOTE: never write a literal
# empty GitHub expression in this run block, even inside a comment — Actions
# evaluates it pre-bash and the whole workflow fails to parse.
USER_CUSTOM_INSTRUCTIONS="$CUSTOM_REPO_INSTRUCTIONS"
# Combine repository context + user custom instructions
# Repository context goes FIRST (highest priority in prompts)
if [ -n "$USER_CUSTOM_INSTRUCTIONS" ]; then
CUSTOM_INSTRUCTIONS="${REPOSITORY_CONTEXT}"$'\n\n'"${USER_CUSTOM_INSTRUCTIONS}"
else
CUSTOM_INSTRUCTIONS="$REPOSITORY_CONTEXT"
fi
# External repos come as separate JSON array parameter
EXTERNAL_REPOS_COUNT=$(echo "$EXTERNAL_REPOS" | jq '. | length' 2>/dev/null || echo "0")
echo "Repository context: ${#REPOSITORY_CONTEXT} chars"
echo "Custom instructions: ${#USER_CUSTOM_INSTRUCTIONS} chars"
echo "Combined prompt block: ${#CUSTOM_INSTRUCTIONS} chars"
echo "External repositories: $EXTERNAL_REPOS_COUNT"
echo "Combined prompt block, first 300 chars:"
echo "${CUSTOM_INSTRUCTIONS:0:300}..."
echo "::endgroup::"
# 3. Parsed values ---------------------------------------------------
# Validate providers
if [[ ! "$CODEWIKI_CLUSTER_PROVIDER" =~ ^(anthropic|openai)$ ]]; then
echo "::error title=Invalid CodeWiki configuration::cluster provider is '$CODEWIKI_CLUSTER_PROVIDER'; expected anthropic or openai."
exit 1
fi
if [[ ! "$CODEWIKI_GENERATION_PROVIDER" =~ ^(anthropic|openai)$ ]]; then
echo "::error title=Invalid CodeWiki configuration::generation provider is '$CODEWIKI_GENERATION_PROVIDER'; expected anthropic or openai."
exit 1
fi
if [[ ! "$CODEWIKI_FALLBACK_PROVIDER" =~ ^(anthropic|openai)$ ]]; then
echo "::error title=Invalid CodeWiki configuration::fallback provider is '$CODEWIKI_FALLBACK_PROVIDER'; expected anthropic or openai."
exit 1
fi
# Model names are not empty
[ -z "$CODEWIKI_CLUSTER_MODEL" ] && echo "::error title=Invalid CodeWiki configuration::cluster model is empty." && exit 1
[ -z "$CODEWIKI_GENERATION_MODEL" ] && echo "::error title=Invalid CodeWiki configuration::generation model is empty." && exit 1
[ -z "$CODEWIKI_FALLBACK_MODEL" ] && echo "::error title=Invalid CodeWiki configuration::fallback model is empty." && exit 1
# Numeric values
[[ ! "$CODEWIKI_CLUSTER_MAX_TOKENS" =~ ^[0-9]+$ ]] && echo "::error title=Invalid CodeWiki configuration::cluster max_tokens is '$CODEWIKI_CLUSTER_MAX_TOKENS', not a number." && exit 1
[[ ! "$CODEWIKI_GENERATION_MAX_TOKENS" =~ ^[0-9]+$ ]] && echo "::error title=Invalid CodeWiki configuration::generation max_tokens is '$CODEWIKI_GENERATION_MAX_TOKENS', not a number." && exit 1
[[ ! "$CODEWIKI_FALLBACK_MAX_TOKENS" =~ ^[0-9]+$ ]] && echo "::error title=Invalid CodeWiki configuration::fallback max_tokens is '$CODEWIKI_FALLBACK_MAX_TOKENS', not a number." && exit 1
[[ ! "$CODEWIKI_MAX_DEPTH" =~ ^[0-9]+$ ]] && echo "::error title=Invalid CodeWiki configuration::max_depth is '$CODEWIKI_MAX_DEPTH', not a number." && exit 1
[[ ! "$CODEWIKI_MAX_FILES_PER_MODULE" =~ ^[0-9]+$ ]] && echo "::error title=Invalid CodeWiki configuration::max_files_per_module is '$CODEWIKI_MAX_FILES_PER_MODULE', not a number." && exit 1
echo "Parsed values: valid"
# 4. Export to GITHUB_ENV (every later step reads these) -------------
# CodeWiki config (31 vars: cluster=9, generation=10, fallback=9, shared=3)
echo "CODEWIKI_CLUSTER_PROVIDER=$CODEWIKI_CLUSTER_PROVIDER" >> $GITHUB_ENV
echo "CODEWIKI_CLUSTER_MODEL=$CODEWIKI_CLUSTER_MODEL" >> $GITHUB_ENV
echo "CODEWIKI_CLUSTER_MAX_TOKENS=$CODEWIKI_CLUSTER_MAX_TOKENS" >> $GITHUB_ENV
echo "CODEWIKI_CLUSTER_MAX_TOKEN_FIELD=$CODEWIKI_CLUSTER_MAX_TOKEN_FIELD" >> $GITHUB_ENV
echo "CODEWIKI_GENERATION_PROVIDER=$CODEWIKI_GENERATION_PROVIDER" >> $GITHUB_ENV
echo "CODEWIKI_GENERATION_MODEL=$CODEWIKI_GENERATION_MODEL" >> $GITHUB_ENV
echo "CODEWIKI_GENERATION_MAX_TOKENS=$CODEWIKI_GENERATION_MAX_TOKENS" >> $GITHUB_ENV
echo "CODEWIKI_GENERATION_MAX_TOKEN_FIELD=$CODEWIKI_GENERATION_MAX_TOKEN_FIELD" >> $GITHUB_ENV
echo "CODEWIKI_FALLBACK_PROVIDER=$CODEWIKI_FALLBACK_PROVIDER" >> $GITHUB_ENV
echo "CODEWIKI_FALLBACK_MODEL=$CODEWIKI_FALLBACK_MODEL" >> $GITHUB_ENV
echo "CODEWIKI_FALLBACK_MAX_TOKENS=$CODEWIKI_FALLBACK_MAX_TOKENS" >> $GITHUB_ENV
echo "CODEWIKI_FALLBACK_MAX_TOKEN_FIELD=$CODEWIKI_FALLBACK_MAX_TOKEN_FIELD" >> $GITHUB_ENV
echo "CODEWIKI_MAX_FILES_PER_MODULE=$CODEWIKI_MAX_FILES_PER_MODULE" >> $GITHUB_ENV
echo "CODEWIKI_MAX_DEPTH=$CODEWIKI_MAX_DEPTH" >> $GITHUB_ENV
echo "CODEWIKI_REPO=$CODEWIKI_REPO" >> $GITHUB_ENV
# YouTube config (2 vars - API key from secrets, not exported here)
echo "YOUTUBE_ENABLED=$YOUTUBE_ENABLED" >> $GITHUB_ENV
echo "YOUTUBE_CHANNELS=$YOUTUBE_CHANNELS" >> $GITHUB_ENV
# README config (3 vars)
echo "README_LOGO_DARK=$README_LOGO_DARK" >> $GITHUB_ENV
echo "README_LOGO_LIGHT=$README_LOGO_LIGHT" >> $GITHUB_ENV
echo "README_LOGO_ALT=$README_LOGO_ALT" >> $GITHUB_ENV
# Output paths (5 vars)
echo "DOCS_OUTPUT_PATH=$DOCS_OUTPUT_PATH" >> $GITHUB_ENV
echo "REFERENCE_OUTPUT_PATH=$REFERENCE_OUTPUT_PATH" >> $GITHUB_ENV
echo "DIAGRAMS_OUTPUT_PATH=$DIAGRAMS_OUTPUT_PATH" >> $GITHUB_ENV
echo "GETTING_STARTED_OUTPUT_PATH=$GETTING_STARTED_OUTPUT_PATH" >> $GITHUB_ENV
echo "DEVELOPMENT_OUTPUT_PATH=$DEVELOPMENT_OUTPUT_PATH" >> $GITHUB_ENV
# Custom instructions (2 vars)
echo "CUSTOM_INSTRUCTIONS<<EOF" >> $GITHUB_ENV
echo "$CUSTOM_INSTRUCTIONS" >> $GITHUB_ENV
echo "EOF" >> $GITHUB_ENV
echo "EXTERNAL_REPOS=$EXTERNAL_REPOS" >> $GITHUB_ENV
echo "Run configuration validated and exported: 15 CodeWiki, 2 YouTube, 3 README, 5 output paths, 2 instruction values"
# =========================================================================
# SOURCE FILE DISCOVERY (the one list every stage reads)
# Discovers SOURCE CODE files and optionally DELETES everything else.
# When SOURCE_FILES_LIMIT > 0:
# 1. Keeps only N source files (.ts, .java, .py, etc.)
# 2. DELETES ALL other files in the repo (aggressive cleanup)
# Generated docs (inline .md) are created AFTER this step, so not affected.
# =========================================================================
- name: Discover source files
id: discover_files
env:
SOURCE_FILES_LIMIT: ${{ env.SOURCE_FILES_LIMIT }}
# DOCS_OUTPUT_PATH is needed below so the find can exclude the
# generated docs tree (deleted by the docs-removal step that runs
# AFTER discovery but BEFORE Stage 1 — source-extension files
# under docs/ would otherwise be enumerated, then deleted, then
# cause ENOENT in Stage 1).
DOCS_OUTPUT_PATH: ${{ env.DOCS_OUTPUT_PATH }}
run: |
source /tmp/workflow-helpers.sh
echo "Discovering source files in the repository (.) and its dependencies (../deps/)"
# File paths (all in /tmp to avoid accidental commits)
SOURCE_FILES_LIST="/tmp/.code-documentation-source-files.txt"
ALL_SOURCE_TEMP="/tmp/all_source_files_discovered.txt"
ALL_FILES_TEMP="/tmp/all_files_in_repo.txt"
FILES_TO_DELETE="/tmp/files_to_delete.txt"
# 1. Find SOURCE CODE files only (with standard exclusions)
# IMPORTANT: exclude $DOCS_OUTPUT_PATH/* — the docs-removal step below
# does `rm -rf $DOCS_OUTPUT_PATH` BEFORE Stage 1 reads this list. If
# any source-extension file lives under the docs tree (Doxygen's
# `docs/doxygen/documentation.h`, Sphinx `_extensions/*.py`, etc.),
# it would be enumerated here, then deleted, then Stage 1 hits
# ENOENT trying to read it. This is the source-discovery /
# docs-removal / Stage-1 ordering bug — exclusion is the targeted fix.
# The list comes from ci-source.mjs: the served source extensions, the
# served never-source directories, tests skipped by name — the same rule
# the language detection above counted with.
node /tmp/ci-source.mjs list "$ALL_SOURCE_TEMP" --exclude "./$DOCS_OUTPUT_PATH" > /dev/null || { echo "::error title=Source discovery failed::ci-source.mjs list exited non-zero."; exit 1; }
TOTAL_SOURCE=$(wc -l < "$ALL_SOURCE_TEMP" | tr -d ' ')
FILE_LIMIT="${SOURCE_FILES_LIMIT:-0}"
echo "Source files found: $TOTAL_SOURCE"
# Apply limit: keep N source files, DELETE EVERYTHING ELSE
if [ "$FILE_LIMIT" -gt 0 ]; then
echo "::warning title=Source file limit active::SOURCE_FILES_LIMIT=$FILE_LIMIT. Keeping $FILE_LIMIT source files and deleting every other file in the checkout (a debugging setting)."
echo "::group::Apply the source file limit"
# Keep first N source files
head -n "$FILE_LIMIT" "$ALL_SOURCE_TEMP" > "$SOURCE_FILES_LIST"
KEEPING=$(wc -l < "$SOURCE_FILES_LIST" | tr -d ' ')
# 2. Find ALL files in the repo (except .git and workflow temp files)
find . ../deps 2>/dev/null -type f \
-not -path "*/.git/*" \
-not -path "*/.git" \
-not -name ".code-documentation-*" \
-not -name ".doc-stage*" \
| sort > "$ALL_FILES_TEMP"
TOTAL_FILES=$(wc -l < "$ALL_FILES_TEMP" | tr -d ' ')
echo "Files in the checkout: $TOTAL_FILES"
# Build delete list: ALL files EXCEPT the ones we're keeping
# Also preserve workflow temp files (.code-documentation-*, .doc-stage*)
> "$FILES_TO_DELETE"
while IFS= read -r file; do
# Skip workflow temp files we need to preserve
case "$file" in
./.code-documentation-*|./.doc-stage*) continue ;;
esac
# Check if this file is in our keep list
if ! grep -qxF "$file" "$SOURCE_FILES_LIST" 2>/dev/null; then
echo "$file" >> "$FILES_TO_DELETE"
fi
done < "$ALL_FILES_TEMP"
DELETE_COUNT=$(wc -l < "$FILES_TO_DELETE" | tr -d ' ')
echo "Files to delete: $DELETE_COUNT"
# Delete all files NOT in the keep list
DELETED_COUNT=0
while IFS= read -r file_to_delete; do
if [ -f "$file_to_delete" ]; then
rm -f "$file_to_delete"
DELETED_COUNT=$((DELETED_COUNT + 1))
fi
done < "$FILES_TO_DELETE"
echo "Kept $KEEPING source files, deleted $DELETED_COUNT files"
rm -f "$FILES_TO_DELETE" "$ALL_FILES_TEMP"
# Aggressively prune directories (including those with only dotfiles)
PRUNED_COUNT=0
# First, delete all dotfiles except in .git and workflow temp files (they prevent dir deletion)
find . -type f -name ".*" \
-not -path "*/.git/*" \
-not -name ".code-documentation-*" \
-not -name ".doc-stage*" \
-delete 2>/dev/null || true
# Multiple passes to handle nested empty directories
for i in 1 2 3 4 5 6 7 8 9 10; do
PASS_COUNT=0
while IFS= read -r empty_dir; do
if [ -d "$empty_dir" ] && [ -z "$(ls -A "$empty_dir" 2>/dev/null)" ]; then
rmdir "$empty_dir" 2>/dev/null && PASS_COUNT=$((PASS_COUNT + 1))
fi
done < <(find . -type d -empty 2>/dev/null | grep -v "^.$" | grep -v ".git")
PRUNED_COUNT=$((PRUNED_COUNT + PASS_COUNT))
[ "$PASS_COUNT" -eq 0 ] && break
done
if [ "$PRUNED_COUNT" -gt 0 ]; then
echo "Pruned $PRUNED_COUNT empty directories"
fi
# Show what's left
echo "Remaining directories (first 20):"
find . -type d -not -path "*/.git/*" -not -path "*/.git" | head -20
echo "::endgroup::"
else
# No limit - keep all source files
cp "$ALL_SOURCE_TEMP" "$SOURCE_FILES_LIST"
fi
# Cleanup temp file
rm -f "$ALL_SOURCE_TEMP"
# Count remaining files
FILE_COUNT=$(count_source_files "$SOURCE_FILES_LIST")
MAIN_COUNT=$(count_main_repo_files "$SOURCE_FILES_LIST")
DEPS_COUNT=$(count_dependency_files "$SOURCE_FILES_LIST")
echo "Source files to document: $FILE_COUNT ($MAIN_COUNT in this repository, $DEPS_COUNT in dependencies)"
echo "::group::Source files by language"
node /tmp/ci-source.mjs breakdown "$SOURCE_FILES_LIST" | sed 's/^/ /'
echo "::endgroup::"
echo "::group::Sample source files (first 20)"
head -20 "$SOURCE_FILES_LIST" | sed 's/^/ /'
echo "::endgroup::"
# Output for use by subsequent steps
set_output "source_file_count" "$FILE_COUNT"
set_output "source_files_list" "$SOURCE_FILES_LIST"
# =========================================================================
# PULL REQUEST FIRST (progressive pull request)
# The branch and the pull request exist before any stage runs, so each
# stage commits its results the moment it finishes. A timeout loses at
# most the stage in flight.
# =========================================================================
- name: Derive branch name
id: branch-name-early
run: |
source /tmp/workflow-helpers.sh
# Replace colons and other invalid chars with hyphens for git branch name
SAFE_RUN_ID=$(echo "$RUN_ID" | sed 's/[:]/-/g' | sed 's/[^a-zA-Z0-9._-]/-/g')
set_output "safe_run_id" "$SAFE_RUN_ID"
echo "Run ID for the branch name: $SAFE_RUN_ID"
- name: Create docs branch
id: create-pr-branch
run: |
source /tmp/workflow-helpers.sh
# Use sanitized run ID for branch name. Branch, PR title, status file and
# commit messages all carry the product name: 🦩 Flamingo Code Documentation.
SAFE_RUN_ID="${{ steps.branch-name-early.outputs.safe_run_id }}"
BRANCH_NAME="${DOCS_BRANCH_PREFIX}$SAFE_RUN_ID"
# Configure git
git config user.name "github-actions[bot]"
git config user.email "github-actions[bot]@users.noreply.github.com"
# Create and push empty branch
git checkout -b "$BRANCH_NAME"
# Create initial commit to enable PR creation
echo "# 🦩 Flamingo Code Documentation: Started" > .flamingo-ai-technical-writer-status.md
echo "" >> .flamingo-ai-technical-writer-status.md
echo "Run ID: $SAFE_RUN_ID" >> .flamingo-ai-technical-writer-status.md
echo "Status: In Progress" >> .flamingo-ai-technical-writer-status.md
echo "Started: $(date -u +"%Y-%m-%d %H:%M:%S UTC")" >> .flamingo-ai-technical-writer-status.md
# -f: the status file is a hidden dot-md that many target repos' .gitignore
# patterns (e.g. `.*` / `*status*`) cover — without -f, `git add` fails the
# step. It's removed again in "Clean up temporary files" before the PR.
git add -f .flamingo-ai-technical-writer-status.md
git commit -m "docs: Initialize 🦩 Flamingo Code Documentation run [skip ci]"
git push -u origin "$BRANCH_NAME"
# Store branch name for later steps
set_output "branch_name" "$BRANCH_NAME"
echo "Docs branch pushed: $BRANCH_NAME"
- name: Ensure pull request labels exist
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: |
# Labels used for 🦩 Flamingo Code Documentation PRs
LABELS=(
"documentation:A label for documentation-related PRs:#0075ca"
"automated:PRs created by automation/bots:#ededed"
"in-progress:Work in progress - not ready for merge:#fbca04"
)
for LABEL_DEF in "${LABELS[@]}"; do
LABEL_NAME=$(echo "$LABEL_DEF" | cut -d: -f1)
LABEL_DESC=$(echo "$LABEL_DEF" | cut -d: -f2)
LABEL_COLOR=$(echo "$LABEL_DEF" | cut -d: -f3 | sed 's/#//')
# Check if label exists
if gh label list --json name --jq '.[].name' | grep -q "^${LABEL_NAME}$"; then
echo "Label exists: $LABEL_NAME"
else
echo "Creating label: $LABEL_NAME"
gh label create "$LABEL_NAME" \
--description "$LABEL_DESC" \
--color "$LABEL_COLOR" || true
fi
done
- name: Open pull request
id: create-initial-pr
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }}
RUN_ID_VAR: ${{ env.RUN_ID }}
REPO_NAME: ${{ github.repository }}
DEFAULT_BRANCH: ${{ github.event.repository.default_branch }}
run: |
source /tmp/workflow-helpers.sh
# Create PR body in a temp file (avoiding YAML parsing issues)
{
echo "🦩 Flamingo Code Documentation: In Progress"
echo ""
echo "Run ID: $RUN_ID_VAR"
echo "Status: Running..."
echo ""
echo "This PR will be updated as each documentation stage completes."
echo ""
echo "Progress"
echo "- Stage 1 Inline Documentation - Starting..."
echo "- Stage 2 Architecture Analysis - Pending"
echo "- Stage 3 Tutorial Generation - Pending"
echo "- Stage 4 Repository Documentation - Pending"
echo ""
echo "Generated by 🦩 Flamingo Code Documentation"
} > /tmp/pr-body.md
# Create PR with gh CLI (works with existing branches)
PR_URL=$(gh pr create \
--base "$DEFAULT_BRANCH" \
--head "$BRANCH_NAME" \
--title "[IN PROGRESS] 🦩 Flamingo Code Documentation" \
--body-file /tmp/pr-body.md \
--label "documentation,automated,in-progress")
# Extract PR number from URL
PR_NUMBER=$(echo "$PR_URL" | grep -oE '[0-9]+$')
echo "Pull request #$PR_NUMBER opened: $PR_URL"
# Set outputs for later steps
set_output "pull-request-url" "$PR_URL"
set_output "pull-request-number" "$PR_NUMBER"
# =========================================================================
# REMOVE DOCS THIS RUN REGENERATES
# Deletes the documentation the configured stages own before they run, so
# the result carries no orphaned files. A subtree whose stage then
# produces nothing is put back by "Restore docs no stage regenerated".
# =========================================================================
- name: Remove docs this run regenerates
env:
DOCS_OUTPUT_PATH: ${{ env.DOCS_OUTPUT_PATH }}
REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }}
DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }}
GETTING_STARTED_OUTPUT_PATH: ${{ env.GETTING_STARTED_OUTPUT_PATH }}
DEVELOPMENT_OUTPUT_PATH: ${{ env.DEVELOPMENT_OUTPUT_PATH }}
BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }}
run: |
source /tmp/workflow-helpers.sh
# Recorded so a stage that ends up producing nothing can restore what was
# deleted on its behalf. Without it, a skipped stage turns the pull request
# into a net DELETION of existing documentation.
PRE_CLEAN_SHA=$(git rev-parse HEAD)
echo "PRE_CLEAN_SHA=$PRE_CLEAN_SHA" >> $GITHUB_ENV
echo "Commit before removal: $PRE_CLEAN_SHA"
# SCOPED TO THE CONFIGURED STAGES.
#
# This used to `rm -rf $DOCS_OUTPUT_PATH` unconditionally. That was safe only
# while every repo ran all four stages. With per-repo `stages`, wiping the
# whole tree when Stage 2 is disabled means the reference architecture and
# diagrams are deleted and never rebuilt — the pull request becomes a net
# DELETION of existing documentation.
CLEAN_TARGETS=()
if [[ "$STAGES" == *"codewiki"* ]]; then
CLEAN_TARGETS+=("$REFERENCE_OUTPUT_PATH" "$DIAGRAMS_OUTPUT_PATH")
fi
if [[ "$STAGES" == *"tutorials"* ]]; then
CLEAN_TARGETS+=("$GETTING_STARTED_OUTPUT_PATH" "$DEVELOPMENT_OUTPUT_PATH")
fi
# DELETE_TARGETS is what actually gets rm -rf'd; CLEAN_TARGETS is what the
# restore step keys PER-STAGE. They differ only for a full wipe: we delete
# the whole tree (so orphaned files from a previous layout — e.g. synthetic
# module_N dirs left by an earlier clustering-failure run — cannot survive)
# but still RECORD the per-stage subtrees, so the restore leaves each
# subtree deleted iff its OWN stage produced output.
#
# Recording the blanket DOCS_OUTPUT_PATH instead (the old behaviour) made the
# restore treat docs/ as a single unit that is "safe to leave deleted" only
# once ALL FOUR stages complete — but the restore runs right after Stage 2,
# so Stage 3/4 are never 'completed' yet, and it restored the ENTIRE pre-clean
# tree every time, undoing the wipe and resurrecting the orphaned module_N docs.
DELETE_TARGETS=("${CLEAN_TARGETS[@]}")
FULL_WIPE=false
SELECTED_COUNT=$(echo "$STAGES" | tr ',' '\n' | grep -c .)
if [ "$SELECTED_COUNT" -ge "${STAGE_COUNT:-4}" ]; then
FULL_WIPE=true
DELETE_TARGETS=("$DOCS_OUTPUT_PATH")
# Record every stage-owned subtree (NOT the blanket docs/) so the restore
# keys each subtree on its own stage instead of the all-four AND.
CLEAN_TARGETS=("$REFERENCE_OUTPUT_PATH" "$DIAGRAMS_OUTPUT_PATH" "$GETTING_STARTED_OUTPUT_PATH" "$DEVELOPMENT_OUTPUT_PATH")
fi
echo "Stages: $STAGES"
echo "Full wipe: $FULL_WIPE"
echo "Targets: ${CLEAN_TARGETS[*]:-(none)}"
# Recorded so the restore step iterates exactly what was removed — the
# stage -> subtree map has ONE home, here.
{
echo "CLEAN_TARGETS_RECORD<<__EOT__"
for t in ${CLEAN_TARGETS[@]+"${CLEAN_TARGETS[@]}"}; do echo "$t"; done
echo "__EOT__"
} >> $GITHUB_ENV
# Count files before deletion (for reporting). Iterate DELETE_TARGETS —
# the actual rm list (blanket docs/ on a full wipe, per-stage subtrees
# otherwise) — not the restore-record CLEAN_TARGETS.
DELETED_FILES=0
for target in ${DELETE_TARGETS[@]+"${DELETE_TARGETS[@]}"}; do
if [ -d "$target" ]; then
TARGET_FILES=$(find "$target" -type f | wc -l | tr -d ' ')
DELETED_FILES=$((DELETED_FILES + TARGET_FILES))
echo "Removing $target ($TARGET_FILES files)"
rm -rf "$target"
else
echo "Skipping $target (does not exist)"
fi
done
if [ ${#DELETE_TARGETS[@]} -eq 0 ]; then
echo "No configured stage owns a docs subtree; nothing to remove"
fi
# Recreate base directory
mkdir -p "$DOCS_OUTPUT_PATH"
# Commit the deletion to git (so it shows in PR)
if [ "$DELETED_FILES" -gt 0 ]; then
git add -A
# Check if there are staged changes
STAGED_COUNT=$(git diff --cached --name-only | wc -l | tr -d ' ')
if [ "$STAGED_COUNT" -gt 0 ]; then
git commit -m "chore(docs): Clean slate - remove all documentation ($DELETED_FILES files) [skip ci]"
git push origin "$BRANCH_NAME"
echo "Committed the removal of $STAGED_COUNT files"
else
echo "Nothing to commit (the docs were already absent)"
fi
fi
- name: Set up Node.js
uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
with:
node-version: '22'
# v5+ caches automatically when it finds a package manager; this job never did.
package-manager-cache: false
# =========================================================================
# STAGE 0: CODE GRAPH (same build as the standalone code-graph job above)
# Tags the SOURCE branch head so the hub can render this run's
# ecosystem.md from the snapshot of the commit being documented. The hub
# promotes to `live` only when the source branch is the default branch.
# Never fatal: a graph failure costs cross-repo facts, not the docs run.
# =========================================================================
# The command is CODE_GRAPH_INSTALL_COMMAND (lib/config/code-graph-workflow.ts).
- name: Install graph dependencies
continue-on-error: true
run: mkdir -p "$RUNNER_TEMP/code-graph-deps" && cd "$RUNNER_TEMP/code-graph-deps" && printf '%s' '{"name":"code-graph-deps","version":"1.0.0","private":true,"dependencies":{"web-tree-sitter":"0.27.0","@vscode/tree-sitter-wasm":"0.3.1","yaml":"2.9.1"}}' > package.json && printf '%s' '{"name":"code-graph-deps","version":"1.0.0","lockfileVersion":3,"requires":true,"packages":{"":{"name":"code-graph-deps","version":"1.0.0","dependencies":{"web-tree-sitter":"0.27.0","@vscode/tree-sitter-wasm":"0.3.1","yaml":"2.9.1"}},"node_modules/web-tree-sitter":{"version":"0.27.0","resolved":"https://registry.npmjs.org/web-tree-sitter/-/web-tree-sitter-0.27.0.tgz","integrity":"sha512-XK08gj6RwTMQatAG7uVRP8MunqotL/XC19vHgkSPKmELgbGPBj4ECvB8haHOUnyj6ls2B8t42UTro14zxGgAHg=="},"node_modules/@vscode/tree-sitter-wasm":{"version":"0.3.1","resolved":"https://registry.npmjs.org/@vscode/tree-sitter-wasm/-/tree-sitter-wasm-0.3.1.tgz","integrity":"sha512-RJFoomET6FajjG511fmQxeBQfU6M24a0aFZPqpid+ttIxanWf1VGytBG0UmsGjt07qmIPJS8U31D+aecuCucsQ=="},"node_modules/yaml":{"version":"2.9.1","resolved":"https://registry.npmjs.org/yaml/-/yaml-2.9.1.tgz","integrity":"sha512-3NxN8+78OdzbT7C/WjGsyfPAtJaN3FNDsWxv7Y7mcDsT/oOmgW8BpyQQFFBnvZE3j9Y2Sdz1ULFLezL7Eb2yFw=="}}}' > package-lock.json && npm ci --ignore-scripts --no-audit --no-fund && echo "CODE_GRAPH_DEPS_DIR=$RUNNER_TEMP/code-graph-deps" >> "$GITHUB_ENV" || { echo "::warning::graph dependencies failed their lockfile-enforced install; continuing without them"; rm -rf "$RUNNER_TEMP/code-graph-deps"; exit 1; }
- name: Build and upload the code graph
id: graph
continue-on-error: true
env:
WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }}
CODE_GRAPH_DEPS_DIR: ${{ env.CODE_GRAPH_DEPS_DIR }}
GITHUB_REPOSITORY: ${{ github.repository }}
CODE_GRAPH_BRANCH: ${{ env.SOURCE_BRANCH }}
CODE_GRAPH_COMMIT_SHA: ${{ env.SOURCE_HEAD_SHA }}
run: node /tmp/code-graph-build.mjs
# =========================================================================
# STAGE 1: INLINE DOCS
# One hidden .md beside each source file, over the discovered file list.
# =========================================================================
- name: Install stage 1 dependencies
if: contains(env.STAGES, 'inline-docs')
# Install generator deps in an ISOLATED tree under RUNNER_TEMP, NOT the target
# repo. npm resolves against an empty package.json here, so a target repo's own
# peer conflicts (e.g. react-accessible-accordion vs react 18) can never make this
# fail. No --legacy-peer-deps / --no-save band-aids. Generators find these via the
# NODE_PATH set on the generate step (RUNNER_TEMP/doc-orch-deps/node_modules).
run: |
mkdir -p "$RUNNER_TEMP/doc-orch-deps" && cd "$RUNNER_TEMP/doc-orch-deps"
npm init -y >/dev/null 2>&1
npm install @anthropic-ai/sdk@0.115.0 zod@3.25.76 glob@13.0.6
- name: Generate inline docs (stage 1)
id: stage1
if: contains(env.STAGES, 'inline-docs')
env:
# SECURITY: Pass secrets per-step with inline masking.
# No ANTHROPIC_API_KEY — this stage calls Claude through the hub
# (/api/ci/claude), which the WEBHOOK_SECRET below authenticates.
WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }}
# Claude model SSOT — see workflow env CLAUDE_MODEL block
CLAUDE_MODEL: ${{ env.CLAUDE_MODEL }}
DOCS_OUTPUT_PATH: ${{ env.DOCS_OUTPUT_PATH }}
# Stage timeout (in hours)
STAGE1_TIMEOUT_HOURS: ${{ env.STAGE1_TIMEOUT_HOURS }}
# Incremental commit + push every N generated docs (see workflow env)
STAGE1_PUSH_INTERVAL: ${{ env.STAGE1_PUSH_INTERVAL }}
BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }}
STAGE1_PUSH_MARKER: /tmp/stage1-progress-pushes
# Unified file discovery result (single source of truth)
SOURCE_FILES_LIST: ${{ steps.discover_files.outputs.source_files_list }}
SOURCE_FILE_COUNT: ${{ steps.discover_files.outputs.source_file_count }}
# NODE_PATH to find modules from /tmp/ scripts
NODE_PATH: ${{ runner.temp }}/doc-orch-deps/node_modules
# Custom AI Instructions (All Stages)
CUSTOM_REPO_INSTRUCTIONS: ${{ env.CUSTOM_INSTRUCTIONS }}
# External Repositories
EXTERNAL_REPOS: ${{ env.EXTERNAL_REPOS }}
run: |
source /tmp/workflow-helpers.sh
echo "Source files: $SOURCE_FILE_COUNT (from $SOURCE_FILES_LIST)"
echo "Progress push: every $STAGE1_PUSH_INTERVAL generated docs, to $BRANCH_NAME"
# Fresh marker: the generator appends one line per progress push, and the
# commit step below reads it to know work was already pushed.
rm -f "$STAGE1_PUSH_MARKER"
# Script already downloaded to /tmp/ in setup step. run_stage records
# the outcome as stage1_status — see its note in workflow-helpers.sh.
run_stage "Stage 1" "$STAGE1_TIMEOUT_HOURS" stage1_status node /tmp/generate-inline-docs.cjs
# Count generated files (hidden .*.md files)
INLINE_DOCS=$(find . -name ".*.md" -newer .git -type f -not -path "./node_modules/*" -not -path "./.git/*" | wc -l)
set_output "stage1_files" "$INLINE_DOCS"
# =========================================================================
# COMMIT STAGE 1 RESULTS (progressive pull request)
# =========================================================================
# `!= ''`, not `== 'completed'`: runs on a FAILED stage too — see run_stage
# in workflow-helpers.sh (partial output is worth committing; the status
# is what reports the truth home). The build gate holds every run_stage
# commit step to this predicate.
- name: Commit and push stage 1 results
if: always() && steps.stage1.outputs.stage1_status != ''
id: commit-stage1
env:
BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }}
STAGE_FILES: ${{ steps.stage1.outputs.stage1_files }}
STAGE1_PUSH_MARKER: /tmp/stage1-progress-pushes
run: |
source /tmp/workflow-helpers.sh
# The generator already commits + pushes every N docs (STAGE1_PUSH_INTERVAL).
# Report how much landed that way; what's left here is the final partial batch.
if [ -s "$STAGE1_PUSH_MARKER" ]; then
PUSHED_BATCHES=$(wc -l < "$STAGE1_PUSH_MARKER" | tr -d ' ')
PUSHED_FILES=$(awk '{ sum += $1 } END { print sum + 0 }' "$STAGE1_PUSH_MARKER")
echo "Already pushed during generation: $PUSHED_FILES files in $PUSHED_BATCHES batches"
fi
# Push any commits the generator made but could not push (transient push failure)
git push origin "HEAD:refs/heads/$BRANCH_NAME" 2>/dev/null || true
# Stage all .md files generated by Stage 1 (hidden inline docs)
find . -name ".*.md" -type f \
-not -path "./node_modules/*" \
-not -path "./.git/*" \
-exec git add -f {} \; 2>/dev/null || true
# Check if there are changes
STAGED_COUNT=$(git diff --cached --name-only | wc -l)
if [ "$STAGED_COUNT" -gt 0 ]; then
# Commit and push
git commit -m "docs: Stage 1 - Inline documentation ($STAGE_FILES files) [skip ci]"
git push origin "$BRANCH_NAME"
echo "Committed and pushed $STAGED_COUNT stage 1 files"
set_output "committed" "true"
elif [ -s "$STAGE1_PUSH_MARKER" ]; then
# Everything already landed via the incremental progress pushes
echo "Nothing left to commit: every stage 1 file was pushed during generation"
set_output "committed" "true"
else
echo "::warning title=Stage 1 produced nothing::No inline docs to commit."
set_output "committed" "false"
fi
# =========================================================================
# ORPHANED INLINE DOCS
# Right after stage 1: removes each hidden .*.md whose source file is gone.
# =========================================================================
# `== 'completed'` HERE IS DELIBERATE, unlike the commit steps: deleting
# "orphaned" docs after a stage that FAILED would delete docs whose
# sources were never re-examined.
- name: Remove orphaned inline docs
if: always() && steps.stage1.outputs.stage1_status == 'completed'
id: orphan-detection-inline
env:
BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }}
run: |
source /tmp/workflow-helpers.sh
# Run orphan detection script
bash /tmp/detect-orphans.sh || true
# Check if orphans were deleted (script writes list to /tmp/orphaned-inline-files.txt)
DELETED_FILES_LIST="/tmp/orphaned-inline-files.txt"
if [ -f "$DELETED_FILES_LIST" ] && [ -s "$DELETED_FILES_LIST" ]; then
DELETED_COUNT=$(wc -l < "$DELETED_FILES_LIST" | tr -d ' ')
echo "Orphaned inline docs removed: $DELETED_COUNT"
# Only add the specific files that were deleted by the script
while IFS= read -r deleted_file; do
git add "$deleted_file" 2>/dev/null || true
done < "$DELETED_FILES_LIST"
# Verify we have staged changes
STAGED_COUNT=$(git diff --cached --name-only | wc -l | tr -d ' ')
if [ "$STAGED_COUNT" -gt 0 ]; then
git commit -m "chore(docs): Remove $DELETED_COUNT orphaned inline files [skip ci]"
git push origin "$BRANCH_NAME"
echo "Committed $STAGED_COUNT orphan removals"
set_output "orphans_deleted" "$STAGED_COUNT"
else
echo "Nothing to commit (the files were already absent)"
set_output "orphans_deleted" "0"
fi
else
echo "No orphaned inline docs"
set_output "orphans_deleted" "0"
fi
- name: Update pull request with stage 1 progress
if: always() && steps.commit-stage1.outputs.committed == 'true' && steps.create-initial-pr.outputs.pull-request-number
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }}
STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files }}
RUN_ID_VAR: ${{ env.RUN_ID }}
REPO_NAME: ${{ github.repository }}
run: |
source /tmp/workflow-helpers.sh
# Update PR body
PR_BODY="## 🦩 Flamingo Code Documentation: In Progress
**Run ID:** \`$RUN_ID_VAR\`
**Status:** 🔄 Running...
This PR is being updated as each documentation stage completes.
### Progress
- ✅ Stage 1: Inline Documentation - Completed ($STAGE1_FILES files)
- ⏳ Stage 2: Architecture Analysis - Running...
- ⏱️ Stage 3: Tutorial Generation - Pending
- ⏱️ Stage 4: Repository Documentation - Pending
---
🦩 Generated by [Flamingo Code Documentation](https://flamingo.run)"
gh pr edit "$PR_NUMBER" --body "$PR_BODY"
echo "Pull request #$PR_NUMBER updated with stage 1 progress"
- name: Report stage 1 progress
if: always() && contains(env.STAGES, 'inline-docs') && env.HUB_BASE_URL != ''
continue-on-error: true
env:
WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }}
WORKFLOW_RUN_ID: ${{ github.run_id }}
WORKFLOW_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }}
STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }}
PR_URL: ${{ steps.create-initial-pr.outputs.pull-request-url }}
PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }}
run: |
source /tmp/workflow-helpers.sh
CALLBACK_URL="${HUB_BASE_URL}/api/code-documentation/webhook"
echo "Reporting stage 1 (inline docs): $STAGE1_STATUS, $STAGE1_FILES files"
report_stage_progress "$CALLBACK_URL" "$WEBHOOK_SECRET" "$RUN_ID" "$REPO_ID" \
"$WORKFLOW_RUN_ID" "$WORKFLOW_URL" "codewiki" \
"$STAGE1_STATUS" "$STAGE1_FILES" "" "0" "" "0" "" "0" \
"$PR_URL" "$PR_NUMBER"
# =========================================================================
# ECOSYSTEM FACTS — derived by the hub from the code graph, never written
# by a model. Two renderings of the same live snapshot: ecosystem.md
# (committed under the reference tree and fed to the Stage 2/3/4 prompts
# as ground truth for the Dependencies sections) and the marker-delimited
# AGENTS.md block (upserted in place, idempotent). A repo with no graph
# yet is a notice, not a failure.
# =========================================================================
- name: Fetch ecosystem facts
continue-on-error: true
env:
WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }}
REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }}
run: |
source /tmp/workflow-helpers.sh
# Through ci-hub.mjs, the shell's way into the ONE hub transport
# (code-review-lib.mjs): it checks the destination before the secret
# leaves, keeps the secret in a header, and writes the body only on a
# 2xx. It prints the status and exits 0 whenever the hub ANSWERED — a 404
# is an answer ("no graph yet"), not a failure.
REPO_PARAM=$(printf '%s' "$GITHUB_REPOSITORY" | sed 's|/|%2F|g')
ECOSYSTEM_PATH="/api/ci/code-graph/ecosystem.md?repo=${REPO_PARAM}"
HTTP_CODE=$(node /tmp/ci-hub.mjs get "$ECOSYSTEM_PATH" /tmp/ecosystem.md) || HTTP_CODE="000"
if [ "$HTTP_CODE" = "404" ]; then
echo "::notice title=No code graph yet::The hub has no code graph for $GITHUB_REPOSITORY yet; the docs are written without ecosystem facts."
rm -f /tmp/ecosystem.md
exit 0
fi
if [ "$HTTP_CODE" != "200" ]; then
echo "::warning title=Ecosystem facts unavailable::The hub answered HTTP $HTTP_CODE; the Dependencies sections are written without cross-repository facts."
rm -f /tmp/ecosystem.md
exit 0
fi
# Kept in /tmp ONLY until Stage 2 has run. The reference directory is
# CodeWiki's output directory, and CodeWiki asks "already contains
# documentation. Overwrite?" when it finds a .md file there — a prompt
# a runner cannot answer, so Stage 2 aborted on every CodeWiki repo
# (CodeWiki run 35293807658). The Stage 2 commit step copies the file
# into place, after either engine has written its own output.
echo "Fetched ecosystem.md ($(wc -c < /tmp/ecosystem.md | tr -d ' ') bytes); it is copied to $REFERENCE_OUTPUT_PATH after stage 2"
HTTP_CODE=$(node /tmp/ci-hub.mjs get "${ECOSYSTEM_PATH}&format=agents" /tmp/ecosystem-agents.md) || HTTP_CODE="000"
if [ "$HTTP_CODE" = "200" ]; then
upsert_marker_block AGENTS.md /tmp/ecosystem-agents.md
# Claude Code reads CLAUDE.md, and AGENTS.md only when a folder has NO CLAUDE.md
# (native fallback since 2.1.277). A repository that keeps a CLAUDE.md therefore gets
# ONE marker-delimited `@AGENTS.md` import, so its agents load the block above (the
# ecosystem facts and the multi-repo change-set rule). Skipped when CLAUDE.md already
# imports AGENTS.md itself; never creates a CLAUDE.md.
IMPORT_START='<!-- flamingo-agents-import:start -->'
IMPORT_END='<!-- flamingo-agents-import:end -->'
if [ -f CLAUDE.md ] && { grep -qF "$IMPORT_START" CLAUDE.md || ! grep -qE '(^|[[:space:]])@AGENTS\.md' CLAUDE.md; }; then
printf '%s\n' "$IMPORT_START" '@AGENTS.md' "$IMPORT_END" > /tmp/claude-agents-import.md
upsert_marker_block CLAUDE.md /tmp/claude-agents-import.md "$IMPORT_START" "$IMPORT_END"
fi
else
echo "::warning title=AGENTS.md block unavailable::The hub answered HTTP $HTTP_CODE; AGENTS.md is left untouched."
rm -f /tmp/ecosystem-agents.md
fi
# =========================================================================
# STAGE 2: REFERENCE DOCS
# Architecture overview, module tree and diagrams. CodeWiki where it can
# parse the primary language ("Detect primary language" decided, before
# stage 1), otherwise the Claude architecture analysis.
# =========================================================================
- name: Set up Python 3.12
if: contains(env.STAGES, 'codewiki') && steps.detect_language.outputs.codewiki_supported == 'true'
uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0
with:
python-version: '3.12'
- name: Install CodeWiki
id: codewiki_install
if: contains(env.STAGES, 'codewiki') && steps.detect_language.outputs.codewiki_supported == 'true'
continue-on-error: true
env:
CODEWIKI_MAX_FILES_PER_MODULE: ${{ env.CODEWIKI_MAX_FILES_PER_MODULE }}
CODEWIKI_REPO: ${{ env.CODEWIKI_REPO }}
run: |
# Install keyrings.alt for headless keyring support in CI environments
# Install ipython to suppress "Mermaidjs magic function not available" warning
# Install colorama for CodeWiki colored terminal output
pip install keyrings.alt ipython colorama
# Clone CodeWiki directly (no pip caching issues)
# Fixes baked into fork:
# - retries=3 for Pydantic AI agents (prevents "Tool exceeded max retries count of 1")
# - Synthetic module creation when clustering returns 0 modules (prevents context overflow)
# - 'children' key fix for synthetic modules
# - module_tree.json path fix (commit c1dfe5c) - loads from base docs dir, not nested module dir
# See: https://github.com/flamingo-stack/CodeWiki
# Extract repo URL from CODEWIKI_REPO (strip git+ prefix and @branch/commit suffix)
REPO_URL=$(echo "$CODEWIKI_REPO" | sed 's|^git+||' | sed 's|@[^@]*$||')
REF=$(echo "$CODEWIKI_REPO" | grep -o '@[^@]*$' | sed 's|^@||' || echo "main")
echo "📦 Cloning CodeWiki from: $REPO_URL (ref: ${REF:-main})"
rm -rf /tmp/CodeWiki
# Clone and checkout - handle both branches and commit hashes
if [[ "${REF}" =~ ^[0-9a-f]{7,40}$ ]]; then
# Commit hash - clone full repo and checkout specific commit
git clone "$REPO_URL" /tmp/CodeWiki
cd /tmp/CodeWiki && git checkout "${REF}" && cd -
else
# Branch name - shallow clone
git clone --depth 1 --branch "${REF:-main}" "$REPO_URL" /tmp/CodeWiki
fi
echo " Commit: $(cd /tmp/CodeWiki && git rev-parse --short HEAD)"
# Install from local clone (reliable, no caching)
echo "📦 Installing CodeWiki from local clone..."
pip install --no-cache-dir /tmp/CodeWiki
source /tmp/workflow-helpers.sh
set_output "codewiki_installed" "true"
- name: Configure CodeWiki
id: codewiki_config
if: contains(env.STAGES, 'codewiki') && steps.detect_language.outputs.codewiki_supported == 'true' && steps.codewiki_install.outputs.codewiki_installed == 'true'
continue-on-error: true
env:
# SECURITY: Pass secrets per-step with inline masking
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
# Set keyring backend via env var (must be set before any keyring operations)
PYTHON_KEYRING_BACKEND: keyrings.alt.file.PlaintextKeyring
# Flamingo Markdown Guidelines path (needed for module import during config/validate)
FLAMINGO_MARKDOWN_GUIDELINES_PATH: /tmp/flamingo-markdown-guidelines.md
# OSS Tenant Structure: Stage 2 outputs (for clean slate deletion)
REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }}
DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }}
run: |
source /tmp/workflow-helpers.sh
# === KEYRING CONFIGURATION FOR CI ===
# CodeWiki stores API keys in system keyring. In CI (no GUI), we must:
# 1. Create keyring config to specify PlaintextKeyring backend
# 2. Create data directory for credential storage
# See: https://github.com/FSoft-AI4Code/CodeWiki - uses keyring.set_password()
echo "🔑 Setting up keyring for headless CI environment..."
# Create keyring configuration directory and config file
mkdir -p ~/.config/python_keyring
cat > ~/.config/python_keyring/keyringrc.cfg << 'KEYRING_CFG'
[backend]
default-keyring=keyrings.alt.file.PlaintextKeyring
KEYRING_CFG
# Ensure keyring data directory exists with proper permissions
mkdir -p ~/.local/share/python_keyring
chmod 700 ~/.local/share/python_keyring
# Debug: Verify keyring is properly configured
echo "📋 Keyring backend verification:"
python3 -c "import keyring; print(f' Active backend: {keyring.get_keyring()}')"
# Configure CodeWiki with separate cluster and generation providers/models
# CodeWiki calls provider APIs directly via --base-url
# Model names should match the provider's API format (no LiteLLM prefix needed)
# OpenAI: gpt-4o, gpt-4-turbo, gpt-4o-mini
# Provider ids come from MODEL_METADATA in lib/constants/ai-models.ts
# See: https://github.com/FSoft-AI4Code/CodeWiki
echo "🔧 Configuring CodeWiki..."
echo " Cluster (Phase 2): $CODEWIKI_CLUSTER_PROVIDER / $CODEWIKI_CLUSTER_MODEL"
echo " Generation (Phase 3+): $CODEWIKI_GENERATION_PROVIDER / $CODEWIKI_GENERATION_MODEL"
# Source helper functions for configuration
source /tmp/workflow-helpers.sh
# Determine API keys for each provider (cluster, generation/main, fallback)
# Each provider can use a different AI service (OpenAI, Anthropic, etc.)
if [ "$CODEWIKI_CLUSTER_PROVIDER" = "anthropic" ]; then
CLUSTER_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}"
echo " Cluster: Using Anthropic API key"
else
CLUSTER_API_KEY="${{ secrets.OPENAI_API_KEY }}"
echo " Cluster: Using OpenAI API key"
fi
if [ "$CODEWIKI_GENERATION_PROVIDER" = "anthropic" ]; then
MAIN_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}"
echo " Generation: Using Anthropic API key"
else
MAIN_API_KEY="${{ secrets.OPENAI_API_KEY }}"
echo " Generation: Using OpenAI API key"
fi
if [ "$CODEWIKI_FALLBACK_PROVIDER" = "anthropic" ]; then
FALLBACK_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}"
echo " Fallback: Using Anthropic API key"
else
FALLBACK_API_KEY="${{ secrets.OPENAI_API_KEY }}"
echo " Fallback: Using OpenAI API key"
fi
# Configure CodeWiki from CODEWIKI_CONFIG_JSON (single extractor: configure_codewiki_from_json)
# Pass per-provider API keys for mixed provider configurations
configure_codewiki_from_json "$CODEWIKI_CONFIG_JSON" "$CLUSTER_API_KEY" "$MAIN_API_KEY" "$FALLBACK_API_KEY"
if [ $? -ne 0 ]; then
echo "❌ CodeWiki configuration failed"
exit 1
fi
# Set environment variables for backward compatibility with run-codewiki-analysis.sh
export MAIN_MODEL="$CODEWIKI_GENERATION_MODEL"
export FALLBACK_MODEL_1="$CODEWIKI_FALLBACK_MODEL"
if [ "$PRIMARY_PROVIDER" = "anthropic" ]; then
export ANTHROPIC_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}"
else
export OPENAI_API_KEY="${{ secrets.OPENAI_API_KEY }}"
fi
# Verify configuration was saved
echo ""
echo "📋 CodeWiki configuration:"
python -m codewiki config show
echo ""
echo "✅ Validating configuration..."
python -m codewiki config validate
echo ""
set_output "codewiki_configured" "true"
- name: Generate reference docs with CodeWiki (stage 2)
id: stage2
if: contains(env.STAGES, 'codewiki') && steps.detect_language.outputs.codewiki_supported == 'true' && steps.codewiki_config.outputs.codewiki_configured == 'true'
continue-on-error: false
env:
# Keyring backend for CI (must match config step)
PYTHON_KEYRING_BACKEND: keyrings.alt.file.PlaintextKeyring
# API keys for both providers (CodeWiki will use the one configured)
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
# OSS Tenant Structure: Stage 2 outputs
REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }}
DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }}
# Stage timeout
STAGE2_TIMEOUT_HOURS: ${{ env.STAGE2_TIMEOUT_HOURS }}
# CodeWiki JSON configuration (required for unified function)
CODEWIKI_CONFIG_JSON: ${{ env.CODEWIKI_CONFIG_JSON }}
# CodeWiki model configuration (cluster, generation, fallback)
CODEWIKI_CLUSTER_PROVIDER: ${{ env.CODEWIKI_CLUSTER_PROVIDER }}
CODEWIKI_CLUSTER_MODEL: ${{ env.CODEWIKI_CLUSTER_MODEL }}
CODEWIKI_CLUSTER_MAX_TOKENS: ${{ env.CODEWIKI_CLUSTER_MAX_TOKENS }}
CODEWIKI_CLUSTER_MAX_TOKEN_FIELD: ${{ env.CODEWIKI_CLUSTER_MAX_TOKEN_FIELD }}
CODEWIKI_GENERATION_PROVIDER: ${{ env.CODEWIKI_GENERATION_PROVIDER }}
CODEWIKI_GENERATION_MODEL: ${{ env.CODEWIKI_GENERATION_MODEL }}
CODEWIKI_GENERATION_MAX_TOKENS: ${{ env.CODEWIKI_GENERATION_MAX_TOKENS }}
CODEWIKI_GENERATION_MAX_TOKEN_FIELD: ${{ env.CODEWIKI_GENERATION_MAX_TOKEN_FIELD }}
CODEWIKI_FALLBACK_PROVIDER: ${{ env.CODEWIKI_FALLBACK_PROVIDER }}
CODEWIKI_FALLBACK_MODEL: ${{ env.CODEWIKI_FALLBACK_MODEL }}
CODEWIKI_FALLBACK_MAX_TOKENS: ${{ env.CODEWIKI_FALLBACK_MAX_TOKENS }}
CODEWIKI_FALLBACK_MAX_TOKEN_FIELD: ${{ env.CODEWIKI_FALLBACK_MAX_TOKEN_FIELD }}
CODEWIKI_MAX_DEPTH: ${{ env.CODEWIKI_MAX_DEPTH }}
# Flamingo Markdown Guidelines path for CodeWiki prompts
FLAMINGO_MARKDOWN_GUIDELINES_PATH: /tmp/flamingo-markdown-guidelines.md
# Markdown Validation Rules (injected into all prompts)
VALIDATION_RULES_PATH: ${{ env.VALIDATION_RULES_PATH }}
# Custom AI Instructions (All Stages)
CUSTOM_REPO_INSTRUCTIONS: ${{ env.CUSTOM_INSTRUCTIONS }}
# External Repositories
EXTERNAL_REPOS: ${{ env.EXTERNAL_REPOS }}
# Dependencies (for CodeWiki multi-path support)
DEPENDENCIES: ${{ env.DEPENDENCIES }}
run: |
# Determine per-provider API keys (same logic as Configure step)
if [ "$CODEWIKI_CLUSTER_PROVIDER" = "anthropic" ]; then
export CLUSTER_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}"
else
export CLUSTER_API_KEY="${{ secrets.OPENAI_API_KEY }}"
fi
if [ "$CODEWIKI_GENERATION_PROVIDER" = "anthropic" ]; then
export MAIN_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}"
else
export MAIN_API_KEY="${{ secrets.OPENAI_API_KEY }}"
fi
if [ "$CODEWIKI_FALLBACK_PROVIDER" = "anthropic" ]; then
export FALLBACK_API_KEY="${{ secrets.ANTHROPIC_API_KEY }}"
else
export FALLBACK_API_KEY="${{ secrets.OPENAI_API_KEY }}"
fi
# Verify dependencies directory before CodeWiki runs
echo ""
echo "🔍 Pre-CodeWiki Dependency Verification:"
echo " Current directory: $(pwd)"
echo " Absolute path: $(realpath .)"
echo ""
if [ -d "./deps" ]; then
echo " ✅ ./deps EXISTS"
echo " Contents: $(ls -1 ./deps 2>/dev/null | wc -l) repositories"
ls -la ./deps 2>/dev/null | head -5
else
echo " ❌ ./deps NOT FOUND"
fi
if [ -d "../deps" ]; then
echo " ✅ ../deps EXISTS"
echo " Absolute: $(realpath ../deps)"
echo " Contents: $(ls -1 ../deps 2>/dev/null | wc -l) repositories"
ls -la ../deps 2>/dev/null | head -5
else
echo " ❌ ../deps NOT FOUND"
fi
echo " DEPENDENCIES env: ${DEPENDENCIES:-<empty>}"
echo ""
# Run externalized CodeWiki analysis script
/tmp/run-codewiki-analysis.sh
# The alternative for any language CodeWiki cannot parse.
- name: Generate reference docs with Claude (stage 2)
id: stage2_alt
if: contains(env.STAGES, 'codewiki') && steps.detect_language.outputs.codewiki_supported == 'false'
env:
# No ANTHROPIC_API_KEY — this stage calls Claude through the hub
# (/api/ci/claude), which the WEBHOOK_SECRET below authenticates.
WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }}
# SSOT — see workflow env CLAUDE_MODEL block
CLAUDE_MODEL: ${{ env.CLAUDE_MODEL }}
PRIMARY_LANGUAGE: ${{ steps.detect_language.outputs.primary_language }}
# OSS Tenant Structure: Stage 2 outputs
REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }}
DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }}
# Unified file discovery result (single source of truth)
SOURCE_FILES_LIST: ${{ steps.discover_files.outputs.source_files_list }}
SOURCE_FILE_COUNT: ${{ steps.discover_files.outputs.source_file_count }}
# Custom AI Instructions (All Stages)
CUSTOM_REPO_INSTRUCTIONS: ${{ env.CUSTOM_INSTRUCTIONS }}
# External Repositories
EXTERNAL_REPOS: ${{ env.EXTERNAL_REPOS }}
# Dependencies (for consistency with other stages)
DEPENDENCIES: ${{ env.DEPENDENCIES }}
run: |
# Run externalized Claude architecture analysis script
/tmp/run-claude-architecture-analysis.sh
# =========================================================================
# RESTORE DOCS NO STAGE REGENERATED
# The docs removal took the reference/diagrams trees because `codewiki` was in
# STAGES. If neither Stage-2 variant then completed — a repo with no
# discoverable source, an install failure, a skip — the deletion would be the
# only Stage-2 change in the pull request, i.e. a net removal of documentation
# nobody asked to remove. Put it back.
# =========================================================================
- name: Restore docs no stage regenerated
if: always()
env:
BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }}
STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status }}
STAGE2_STATUS: ${{ steps.stage2.outputs.stage2_status }}
STAGE2_ALT_STATUS: ${{ steps.stage2_alt.outputs.stage2_status }}
run: |
# The docs removal deleted whatever the configured stages own, and committed that
# deletion. Any owned subtree whose stage then produced nothing must be put
# back — otherwise the pull request is a net REMOVAL of documentation nobody
# asked to remove. Iterates the exact list the removal recorded, so the
# stage -> subtree map has one home.
if [ -z "$CLEAN_TARGETS_RECORD" ]; then
echo "Nothing was removed; nothing to restore"
exit 0
fi
STAGE2_OK=false
[ "$STAGE2_STATUS" = "completed" ] && STAGE2_OK=true
[ "$STAGE2_ALT_STATUS" = "completed" ] && STAGE2_OK=true
# CodeWiki leaves multi-GB scratch behind on a failed run; never let it near
# the index.
rm -rf "$REFERENCE_OUTPUT_PATH/temp" "$DIAGRAMS_OUTPUT_PATH/temp" 2>/dev/null || true
RESTORED=0
while IFS= read -r target; do
[ -n "$target" ] || continue
# A target is safe to leave deleted only if something regenerated it.
case "$target" in
"$REFERENCE_OUTPUT_PATH"|"$DIAGRAMS_OUTPUT_PATH")
[ "$STAGE2_OK" = true ] && continue ;;
"$GETTING_STARTED_OUTPUT_PATH"|"$DEVELOPMENT_OUTPUT_PATH")
[ "$STAGE3_STATUS" = "completed" ] && continue ;;
"$DOCS_OUTPUT_PATH")
# Full wipe: only fully safe when every stage delivered.
if [ "$STAGE1_STATUS" = "completed" ] && [ "$STAGE2_OK" = true ] && \
[ "$STAGE3_STATUS" = "completed" ] && [ "$STAGE4_STATUS" = "completed" ]; then
continue
fi ;;
esac
if git checkout "$PRE_CLEAN_SHA" -- "$target" 2>/dev/null; then
echo "::notice title=Docs restored::$target restored from $PRE_CLEAN_SHA; no stage regenerated it."
RESTORED=1
fi
done <<< "$CLEAN_TARGETS_RECORD"
# `git checkout -- <path>` already stages exactly those paths. Deliberately NO
# `git add -A`: at this point the workspace holds npm install output from
# Stage 1 and, on a failed CodeWiki run, its scratch trees.
if [ "$RESTORED" -eq 1 ] && [ -n "$(git diff --cached --name-only)" ]; then
git commit -m "chore(docs): restore documentation no stage regenerated [skip ci]"
git push origin "$BRANCH_NAME"
echo "Restore committed and pushed"
else
echo "Nothing to restore"
fi
# =========================================================================
# COMMIT STAGE 2 RESULTS (progressive pull request)
# =========================================================================
- name: Commit and push stage 2 results
if: always() && (steps.stage2.outputs.stage2_status == 'completed' || steps.stage2_alt.outputs.stage2_status == 'completed')
id: commit-stage2
env:
BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }}
STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files }}
REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }}
DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }}
run: |
source /tmp/workflow-helpers.sh
# Remove CodeWiki scratch from every output directory
rm -rf "$REFERENCE_OUTPUT_PATH/temp" 2>/dev/null || echo "::warning::Could not remove $REFERENCE_OUTPUT_PATH/temp"
rm -rf "$DIAGRAMS_OUTPUT_PATH/temp" 2>/dev/null || echo "::warning::Could not remove $DIAGRAMS_OUTPUT_PATH/temp"
# Verify cleanup
if [ -d "$REFERENCE_OUTPUT_PATH/temp" ]; then
echo "::error title=Scratch not removed::$REFERENCE_OUTPUT_PATH/temp is still present; refusing to commit it."
ls -la "$REFERENCE_OUTPUT_PATH/temp"
exit 1
fi
# .gitignore in each output directory keeps CodeWiki scratch out of commits
for output_dir in "$REFERENCE_OUTPUT_PATH" "$DIAGRAMS_OUTPUT_PATH"; do
if [ -d "$output_dir" ]; then
{
echo "# CodeWiki temp files (dependency graphs can be 7GB+)"
echo "temp/"
echo "dependency_graphs/"
echo ""
echo "# JSON intermediate files (except schema/config)"
echo "*.json"
echo "!*-schema.json"
echo "!*-config.json"
} > "$output_dir/.gitignore"
echo "Wrote $output_dir/.gitignore"
fi
done
echo "::group::Stage 2 output before staging"
echo "Reference directory ($REFERENCE_OUTPUT_PATH):"
if [ -d "$REFERENCE_OUTPUT_PATH" ]; then
find "$REFERENCE_OUTPUT_PATH" -type f \( -name "*.md" -o -name "*.mmd" -o -name ".gitignore" \) | head -20
FILE_COUNT=$(find "$REFERENCE_OUTPUT_PATH" -type f \( -name "*.md" -o -name "*.mmd" \) | wc -l | tr -d ' ')
echo ".md/.mmd files: $FILE_COUNT"
else
echo "(directory does not exist)"
fi
echo "Diagrams directory ($DIAGRAMS_OUTPUT_PATH):"
if [ -d "$DIAGRAMS_OUTPUT_PATH" ]; then
find "$DIAGRAMS_OUTPUT_PATH" -type f \( -name "*.md" -o -name "*.mmd" -o -name ".gitignore" \) | head -20
FILE_COUNT=$(find "$DIAGRAMS_OUTPUT_PATH" -type f \( -name "*.md" -o -name "*.mmd" \) | wc -l | tr -d ' ')
echo ".md/.mmd files: $FILE_COUNT"
else
echo "(directory does not exist)"
fi
echo "::endgroup::"
# The hub-rendered ecosystem.md joins the reference directory only NOW,
# after the engine has run: placed earlier it makes CodeWiki prompt for
# an overwrite (see "Fetch ecosystem facts").
if [ -f /tmp/ecosystem.md ]; then
mkdir -p "$REFERENCE_OUTPUT_PATH"
cp /tmp/ecosystem.md "$REFERENCE_OUTPUT_PATH/ecosystem.md"
echo "Added ecosystem.md to $REFERENCE_OUTPUT_PATH/"
fi
# Stage all .md, .mmd, and .gitignore files from Stage 2 output directories
# CRITICAL: Use git add on full paths to preserve nested directory structure
# This ensures Backend/Authentication/JWT/JWT.md keeps its full path in git
# Add only .md, .mmd, .gitignore, and allowed JSON files (*-schema.json, *-config.json)
# This excludes CodeWiki intermediate files: module_tree.json, first_module_tree.json, metadata.json
if [ -d "$REFERENCE_OUTPUT_PATH" ]; then
find "$REFERENCE_OUTPUT_PATH" -type f \( \
-name "*.md" -o -name "*.mmd" -o -name ".gitignore" \
-o -name "*-schema.json" -o -name "*-config.json" \
\) -exec git add {} \; 2>/dev/null || true
echo "Staged .md/.mmd/.gitignore/*-schema.json/*-config.json under $REFERENCE_OUTPUT_PATH/"
fi
if [ -d "$DIAGRAMS_OUTPUT_PATH" ]; then
find "$DIAGRAMS_OUTPUT_PATH" -type f \( \
-name "*.md" -o -name "*.mmd" -o -name ".gitignore" \
-o -name "*-schema.json" -o -name "*-config.json" \
\) -exec git add {} \; 2>/dev/null || true
echo "Staged .md/.mmd/.gitignore/*-schema.json/*-config.json under $DIAGRAMS_OUTPUT_PATH/"
fi
# AGENTS.md carries the hub-rendered ecosystem block ("Fetch ecosystem
# facts" upserts it just before this stage). It sits at the repository
# root, outside every output directory staged above, so it is staged by
# name: the first production run wrote the block and no commit ever
# picked the file up (openframe-cli#383).
if [ -f AGENTS.md ]; then
git add -f AGENTS.md
echo "Staged AGENTS.md (ecosystem block)"
fi
# CLAUDE.md carries the `@AGENTS.md` import the same step keeps (only when it changed).
if [ -f CLAUDE.md ] && ! git diff --quiet -- CLAUDE.md; then
git add -f CLAUDE.md
echo " Added CLAUDE.md (@AGENTS.md import)"
fi
# Check if there are changes
STAGED_COUNT=$(git diff --cached --name-only | wc -l)
echo "Staged files: $STAGED_COUNT"
if [ "$STAGED_COUNT" -gt 0 ]; then
echo "::group::Staged stage 2 files (first 30)"
git diff --cached --name-only | head -30
echo "::endgroup::"
fi
if [ "$STAGED_COUNT" -gt 0 ]; then
# Commit and push
git commit -m "docs: Stage 2 - Architecture analysis ($STAGE2_FILES files) [skip ci]"
git push origin "$BRANCH_NAME"
echo "Committed and pushed $STAGED_COUNT stage 2 files"
set_output "committed" "true"
else
echo "::error title=Stage 2 output not staged::Stage 2 reported $STAGE2_FILES files but none were staged; the files were generated outside the reference and diagrams directories, or not at all."
set_output "committed" "false"
fi
- name: Update pull request with stage 2 progress
if: always() && steps.commit-stage2.outputs.committed == 'true' && steps.create-initial-pr.outputs.pull-request-number
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }}
STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }}
STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }}
RUN_ID_VAR: ${{ env.RUN_ID }}
REPO_NAME: ${{ github.repository }}
run: |
source /tmp/workflow-helpers.sh
# Update PR body
PR_BODY="## 🦩 Flamingo Code Documentation: In Progress
**Run ID:** \`$RUN_ID_VAR\`
**Status:** 🔄 Running...
This PR is being updated as each documentation stage completes.
### Progress
- ✅ Stage 1: Inline Documentation - Completed ($STAGE1_FILES files)
- ✅ Stage 2: Architecture Analysis - Completed ($STAGE2_FILES files)
- ⏳ Stage 3: Tutorial Generation - Running...
- ⏱️ Stage 4: Repository Documentation - Pending
---
🦩 Generated by [Flamingo Code Documentation](https://flamingo.run)"
gh pr edit "$PR_NUMBER" --body "$PR_BODY"
echo "Pull request #$PR_NUMBER updated with stage 2 progress"
- name: Report stage 2 progress
if: always() && contains(env.STAGES, 'codewiki') && env.HUB_BASE_URL != ''
continue-on-error: true
env:
WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }}
WORKFLOW_RUN_ID: ${{ github.run_id }}
WORKFLOW_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }}
STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }}
# Use outputs from either CodeWiki (stage2) or Claude alternative (stage2_alt)
STAGE2_STATUS: ${{ steps.stage2.outputs.stage2_status || steps.stage2_alt.outputs.stage2_status || 'skipped' }}
STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }}
PR_URL: ${{ steps.create-initial-pr.outputs.pull-request-url }}
PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }}
run: |
source /tmp/workflow-helpers.sh
CALLBACK_URL="${HUB_BASE_URL}/api/code-documentation/webhook"
echo "Reporting stage 2 (reference docs): $STAGE2_STATUS, $STAGE2_FILES files"
report_stage_progress "$CALLBACK_URL" "$WEBHOOK_SECRET" "$RUN_ID" "$REPO_ID" \
"$WORKFLOW_RUN_ID" "$WORKFLOW_URL" "tutorials" \
"$STAGE1_STATUS" "$STAGE1_FILES" "$STAGE2_STATUS" "$STAGE2_FILES" "" "0" "" "0" \
"$PR_URL" "$PR_NUMBER"
# =========================================================================
# STAGE 3: TUTORIALS
# Getting-started guides and how-to tutorials, written by the Code
# Documentation lane (code-documentation-lib.mjs): one hub call per
# tutorial with the forced `emit_document` tool and, once the repository's
# visibility is resolved, the hub's read tools (the code graph and the
# rules, scoped to what THIS repository may see). No model key here.
# Generates 4 tutorials: user/getting-started, user/common-use-cases,
# dev/getting-started-dev, dev/architecture-overview-dev
# =========================================================================
- name: Install tutorial and repository docs dependencies
# Stage 3 AND Stage 4 (generate-repo-docs.cjs) share this tree. Gated
# on either: a repository configured with inline-docs + repo-docs and no
# tutorials reached Stage 4 with no dependencies at all, run 35302244804.
if: contains(env.STAGES, 'tutorials') || contains(env.STAGES, 'repo-docs')
# Isolated deps tree (see Stage 1) - no reconciliation with the target repo.
# No model SDK: both stages call Claude through the hub.
run: |
mkdir -p "$RUNNER_TEMP/doc-orch-deps" && cd "$RUNNER_TEMP/doc-orch-deps"
npm init -y >/dev/null 2>&1
npm install zod@3.25.76 glob@13.0.6
- name: Generate tutorials (stage 3)
id: stage3
if: contains(env.STAGES, 'tutorials')
env:
# No ANTHROPIC_API_KEY — this stage calls Claude through the hub
# (/api/ci/claude), which the WEBHOOK_SECRET below authenticates.
WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }}
# Pass through output paths from workflow env (OSS Tenant Structure)
DOCS_OUTPUT_PATH: ${{ env.DOCS_OUTPUT_PATH }}
# Stage 2 outputs (for context)
REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }}
DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }}
# Stage 3 outputs
GETTING_STARTED_OUTPUT_PATH: ${{ env.GETTING_STARTED_OUTPUT_PATH }}
DEVELOPMENT_OUTPUT_PATH: ${{ env.DEVELOPMENT_OUTPUT_PATH }}
# Claude model SSOT — see workflow env CLAUDE_MODEL block
CLAUDE_MODEL: ${{ env.CLAUDE_MODEL }}
# Stage timeout
STAGE3_TIMEOUT_HOURS: ${{ env.STAGE3_TIMEOUT_HOURS }}
# Unified file discovery result (same files as Stage 1 and 2)
SOURCE_FILES_LIST: ${{ steps.discover_files.outputs.source_files_list }}
# NODE_PATH to find modules from /tmp/ scripts
NODE_PATH: ${{ runner.temp }}/doc-orch-deps/node_modules
# YouTube Integration - SECURITY: API key from secrets, NOT dispatch payload
YOUTUBE_ENABLED: ${{ env.YOUTUBE_ENABLED }}
YOUTUBE_CHANNELS: ${{ env.YOUTUBE_CHANNELS }}
YOUTUBE_API_KEY: ${{ secrets.YOUTUBE_API_KEY }}
# Markdown Validation Rules (injected into prompts)
VALIDATION_RULES_PATH: ${{ env.VALIDATION_RULES_PATH }}
# Flamingo Markdown Guidelines (optional)
GUIDELINES_PATH: ${{ env.GUIDELINES_PATH }}
# Stage 3 tracking files (configurable paths)
STAGE3_FILES_TRACKER: ${{ env.STAGE3_FILES_TRACKER }}
STAGE3_STATS_FILE: ${{ env.STAGE3_STATS_FILE }}
# Custom AI Instructions (All Stages)
CUSTOM_REPO_INSTRUCTIONS: ${{ env.CUSTOM_INSTRUCTIONS }}
# Analysis Exclusions
EXCLUDED_PATHS: ${{ env.EXCLUDED_PATHS }}
# External Repositories
EXTERNAL_REPOS: ${{ env.EXTERNAL_REPOS }}
run: |
source /tmp/workflow-helpers.sh
# Script already downloaded to /tmp/ in setup step. run_stage records
# the outcome as stage3_status — see its note in workflow-helpers.sh.
run_stage "Stage 3" "$STAGE3_TIMEOUT_HOURS" stage3_status node /tmp/generate-tutorials-voltagent.cjs
# Count files from both OSS Tenant Structure directories
GETTING_STARTED_FILES=$(count_markdown_files "${GETTING_STARTED_OUTPUT_PATH}")
DEVELOPMENT_FILES=$(count_markdown_files "${DEVELOPMENT_OUTPUT_PATH}")
TUTORIAL_FILES=$((GETTING_STARTED_FILES + DEVELOPMENT_FILES))
echo "Tutorials written: $TUTORIAL_FILES ($GETTING_STARTED_FILES getting started, $DEVELOPMENT_FILES development)"
set_output "stage3_files" "$TUTORIAL_FILES"
# =========================================================================
# COMMIT STAGE 3 RESULTS (progressive pull request)
# =========================================================================
# `!= ''` — the run_stage commit rule, stated once at Stage 1.
- name: Commit and push stage 3 results
if: always() && steps.stage3.outputs.stage3_status != ''
id: commit-stage3
env:
BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }}
STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files }}
GETTING_STARTED_OUTPUT_PATH: ${{ env.GETTING_STARTED_OUTPUT_PATH }}
DEVELOPMENT_OUTPUT_PATH: ${{ env.DEVELOPMENT_OUTPUT_PATH }}
run: |
source /tmp/workflow-helpers.sh
# .gitignore in each tutorial output directory
for output_dir in "$GETTING_STARTED_OUTPUT_PATH" "$DEVELOPMENT_OUTPUT_PATH"; do
if [ -d "$output_dir" ]; then
{
echo "# VoltAgent temp files"
echo "temp/"
echo ""
echo "# JSON intermediate files (except schema/config)"
echo "*.json"
echo "!*-schema.json"
echo "!*-config.json"
} > "$output_dir/.gitignore"
echo "Wrote $output_dir/.gitignore"
fi
done
# Stage all .md and .gitignore files from Stage 3 output directories
find "$GETTING_STARTED_OUTPUT_PATH" "$DEVELOPMENT_OUTPUT_PATH" -type f \( -name "*.md" -o -name ".gitignore" \) \
-exec git add -f {} \; 2>/dev/null || true
# Check if there are changes
STAGED_COUNT=$(git diff --cached --name-only | wc -l)
if [ "$STAGED_COUNT" -gt 0 ]; then
# Commit and push
git commit -m "docs: Stage 3 - Tutorial generation ($STAGE3_FILES files) [skip ci]"
git push origin "$BRANCH_NAME"
echo "Committed and pushed $STAGED_COUNT stage 3 files"
set_output "committed" "true"
else
echo "::warning title=Stage 3 produced nothing::No tutorials to commit."
set_output "committed" "false"
fi
- name: Update pull request with stage 3 progress
if: always() && steps.commit-stage3.outputs.committed == 'true' && steps.create-initial-pr.outputs.pull-request-number
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }}
STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }}
STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }}
STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || 0 }}
RUN_ID_VAR: ${{ env.RUN_ID }}
REPO_NAME: ${{ github.repository }}
run: |
source /tmp/workflow-helpers.sh
# Update PR body
PR_BODY="## 🦩 Flamingo Code Documentation: In Progress
**Run ID:** \`$RUN_ID_VAR\`
**Status:** 🔄 Running...
This PR is being updated as each documentation stage completes.
### Progress
- ✅ Stage 1: Inline Documentation - Completed ($STAGE1_FILES files)
- ✅ Stage 2: Architecture Analysis - Completed ($STAGE2_FILES files)
- ✅ Stage 3: Tutorial Generation - Completed ($STAGE3_FILES files)
- ⏳ Stage 4: Repository Documentation - Running...
---
🦩 Generated by [Flamingo Code Documentation](https://flamingo.run)"
gh pr edit "$PR_NUMBER" --body "$PR_BODY"
echo "Pull request #$PR_NUMBER updated with stage 3 progress"
- name: Report stage 3 progress
if: always() && contains(env.STAGES, 'tutorials') && env.HUB_BASE_URL != ''
continue-on-error: true
env:
WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }}
WORKFLOW_RUN_ID: ${{ github.run_id }}
WORKFLOW_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }}
STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }}
STAGE2_STATUS: ${{ steps.stage2.outputs.stage2_status || steps.stage2_alt.outputs.stage2_status || 'skipped' }}
STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }}
STAGE3_STATUS: ${{ steps.stage3.outputs.stage3_status || 'skipped' }}
STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || 0 }}
PR_URL: ${{ steps.create-initial-pr.outputs.pull-request-url }}
PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }}
run: |
source /tmp/workflow-helpers.sh
CALLBACK_URL="${HUB_BASE_URL}/api/code-documentation/webhook"
echo "Reporting stage 3 (tutorials): $STAGE3_STATUS, $STAGE3_FILES files"
report_stage_progress "$CALLBACK_URL" "$WEBHOOK_SECRET" "$RUN_ID" "$REPO_ID" \
"$WORKFLOW_RUN_ID" "$WORKFLOW_URL" "repo-docs" \
"$STAGE1_STATUS" "$STAGE1_FILES" "$STAGE2_STATUS" "$STAGE2_FILES" "$STAGE3_STATUS" "$STAGE3_FILES" "" "0" \
"$PR_URL" "$PR_NUMBER"
# =========================================================================
# STAGE 4: REPOSITORY DOCS
# Copies LICENSE.md, SECURITY.md from template repo
# Generates/updates README.md, CONTRIBUTING.md and the docs index through
# the Code Documentation lane (one hub call per document, no model key)
# =========================================================================
- name: Generate repository docs (stage 4)
id: stage4
if: contains(env.STAGES, 'repo-docs')
env:
# No ANTHROPIC_API_KEY — this stage calls Claude through the hub.
WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }}
TEMPLATE_REPO: ${{ env.TEMPLATE_REPO }}
TEMPLATE_BRANCH: ${{ env.TEMPLATE_BRANCH }}
DOCS_OUTPUT_PATH: ${{ env.DOCS_OUTPUT_PATH }}
# OSS Tenant Structure: All output paths for docs/README.md navigation
REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }}
DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }}
GETTING_STARTED_OUTPUT_PATH: ${{ env.GETTING_STARTED_OUTPUT_PATH }}
DEVELOPMENT_OUTPUT_PATH: ${{ env.DEVELOPMENT_OUTPUT_PATH }}
# Claude model SSOT — see workflow env CLAUDE_MODEL block
CLAUDE_MODEL: ${{ env.CLAUDE_MODEL }}
STAGE4_TIMEOUT_HOURS: ${{ env.STAGE4_TIMEOUT_HOURS }}
# NODE_PATH to find modules from /tmp/ scripts
NODE_PATH: ${{ runner.temp }}/doc-orch-deps/node_modules
# YouTube Integration - SECURITY: API key from secrets, NOT dispatch payload
YOUTUBE_ENABLED: ${{ env.YOUTUBE_ENABLED }}
YOUTUBE_CHANNELS: ${{ env.YOUTUBE_CHANNELS }}
YOUTUBE_API_KEY: ${{ secrets.YOUTUBE_API_KEY }}
# Markdown Validation Rules (injected into prompts)
VALIDATION_RULES_PATH: ${{ env.VALIDATION_RULES_PATH }}
# Flamingo Markdown Guidelines (optional)
GUIDELINES_PATH: ${{ env.GUIDELINES_PATH }}
# Stage 4 tracking files (configurable paths)
STAGE4_FILES_TRACKER: ${{ env.STAGE4_FILES_TRACKER }}
# Custom AI Instructions (All Stages)
CUSTOM_REPO_INSTRUCTIONS: ${{ env.CUSTOM_INSTRUCTIONS }}
# Analysis Exclusions
EXCLUDED_PATHS: ${{ env.EXCLUDED_PATHS }}
# External Repositories
EXTERNAL_REPOS: ${{ env.EXTERNAL_REPOS }}
# README Branding
README_LOGO_DARK: ${{ env.README_LOGO_DARK }}
README_LOGO_LIGHT: ${{ env.README_LOGO_LIGHT }}
README_LOGO_ALT: ${{ env.README_LOGO_ALT }}
run: |
source /tmp/workflow-helpers.sh
echo "Template repository: $TEMPLATE_REPO@$TEMPLATE_BRANCH"
RAW_URL="https://raw.githubusercontent.com/$TEMPLATE_REPO/$TEMPLATE_BRANCH"
# 1. LICENSE.md and SECURITY.md from the template repository
if curl -fsSL "$RAW_URL/LICENSE.md" -o LICENSE.md 2>/dev/null; then
echo "Copied LICENSE.md from the template repository"
else
echo "::notice title=LICENSE.md not copied::The template repository $TEMPLATE_REPO@$TEMPLATE_BRANCH has no LICENSE.md."
fi
if curl -fsSL "$RAW_URL/SECURITY.md" -o SECURITY.md 2>/dev/null; then
echo "Copied SECURITY.md from the template repository"
else
echo "::notice title=SECURITY.md not copied::The template repository $TEMPLATE_REPO@$TEMPLATE_BRANCH has no SECURITY.md."
fi
# 2. The existing README is context for the new one
if [ -f "README.md" ]; then
README_SIZE=$(wc -c < README.md | tr -d ' ')
echo "Existing README.md: $README_SIZE bytes (used as context)"
else
echo "Existing README.md: none"
fi
# 3. README, CONTRIBUTING and the docs index, one hub call each.
# Script already downloaded to /tmp/ in "Download pipeline scripts"
# run_stage records the outcome as stage4_status — see its note in
# workflow-helpers.sh. The file count below is REPORTING, not a
# status: inferring "completed" from it meant a crashed run that left
# a previous commit's README standing reported success.
run_stage "Stage 4" "$STAGE4_TIMEOUT_HOURS" stage4_status node /tmp/generate-repo-docs.cjs
# 4. Count results
REPO_DOCS=0
for f in README.md CONTRIBUTING.md LICENSE.md SECURITY.md; do
if [ -f "$f" ]; then
SIZE=$(wc -c < "$f" | tr -d ' ')
echo "$f: $SIZE bytes"
REPO_DOCS=$((REPO_DOCS + 1))
fi
done
set_output "stage4_files" "$REPO_DOCS"
if [ "$REPO_DOCS" -gt 0 ]; then
echo "Repository docs present: $REPO_DOCS"
else
echo "::warning title=Stage 4 produced nothing::No repository docs (README.md, CONTRIBUTING.md, LICENSE.md, SECURITY.md) are present."
fi
# =========================================================================
# COMMIT STAGE 4 RESULTS (progressive pull request)
# =========================================================================
# `!= ''` — the run_stage commit rule, stated once at Stage 1.
- name: Commit and push stage 4 results
if: always() && steps.stage4.outputs.stage4_status != ''
id: commit-stage4
env:
BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }}
STAGE4_FILES: ${{ steps.stage4.outputs.stage4_files }}
run: |
source /tmp/workflow-helpers.sh
# .gitignore in each stage 4 managed directory
for managed_dir in docs/api docs/deployment docs/operations docs/cli; do
if [ -d "$managed_dir" ]; then
{
echo "# VoltAgent temp files"
echo "temp/"
echo ""
echo "# JSON intermediate files (except schema/config)"
echo "*.json"
echo "!*-schema.json"
echo "!*-config.json"
} > "$managed_dir/.gitignore"
echo "Wrote $managed_dir/.gitignore"
fi
done
# Stage repository documentation files
# AGENTS.md (and CLAUDE.md's @AGENTS.md import) are here as the backstop for a run whose Stage 2 commit did not happen.
for f in README.md CONTRIBUTING.md LICENSE.md SECURITY.md AGENTS.md CLAUDE.md; do
if [ -f "$f" ]; then
git add -f "$f"
fi
done
# Stage Stage 4 managed directories
for managed_dir in docs/api docs/deployment docs/operations docs/cli; do
if [ -d "$managed_dir" ]; then
git add -f "$managed_dir/" 2>/dev/null || true
fi
done
# Stage docs/README.md if exists
if [ -f "docs/README.md" ]; then
git add -f "docs/README.md"
fi
# Check if there are changes
STAGED_COUNT=$(git diff --cached --name-only | wc -l)
if [ "$STAGED_COUNT" -gt 0 ]; then
# Commit and push
git commit -m "docs: Stage 4 - Repository documentation ($STAGE4_FILES files) [skip ci]"
git push origin "$BRANCH_NAME"
echo "Committed and pushed $STAGED_COUNT stage 4 files"
set_output "committed" "true"
else
echo "No stage 4 changes to commit"
set_output "committed" "false"
fi
- name: Update pull request with stage 4 progress
if: always() && steps.commit-stage4.outputs.committed == 'true' && steps.create-initial-pr.outputs.pull-request-number
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }}
STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }}
STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }}
STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || 0 }}
STAGE4_FILES: ${{ steps.stage4.outputs.stage4_files || 0 }}
RUN_ID_VAR: ${{ env.RUN_ID }}
REPO_NAME: ${{ github.repository }}
run: |
source /tmp/workflow-helpers.sh
# Update PR body
PR_BODY="## 🦩 Flamingo Code Documentation: In Progress
**Run ID:** \`$RUN_ID_VAR\`
**Status:** 🔄 Running...
This PR is being updated as each documentation stage completes.
### Progress
- ✅ Stage 1: Inline Documentation - Completed ($STAGE1_FILES files)
- ✅ Stage 2: Architecture Analysis - Completed ($STAGE2_FILES files)
- ✅ Stage 3: Tutorial Generation - Completed ($STAGE3_FILES files)
- ✅ Stage 4: Repository Documentation - Completed ($STAGE4_FILES files)
---
🦩 Generated by [Flamingo Code Documentation](https://flamingo.run)"
gh pr edit "$PR_NUMBER" --body "$PR_BODY"
echo "Pull request #$PR_NUMBER updated with stage 4 progress"
# =========================================================================
# VALIDATE GENERATED MARKDOWN
# Warn-only validation (never blocks the pull request)
# =========================================================================
- name: Validate generated Markdown
if: always()
continue-on-error: true # NEVER block PR - validation is warn-only
env:
DOCS_OUTPUT_DIR: ${{ env.DOCS_OUTPUT_PATH }}
# OSS Tenant Structure paths for validation
REFERENCE_OUTPUT_PATH: ${{ env.REFERENCE_OUTPUT_PATH }}
DIAGRAMS_OUTPUT_PATH: ${{ env.DIAGRAMS_OUTPUT_PATH }}
GETTING_STARTED_OUTPUT_PATH: ${{ env.GETTING_STARTED_OUTPUT_PATH }}
DEVELOPMENT_OUTPUT_PATH: ${{ env.DEVELOPMENT_OUTPUT_PATH }}
NODE_PATH: ${{ runner.temp }}/doc-orch-deps/node_modules
run: |
# Warn-only: findings are logged per directory and never fail the run.
echo "::group::Validate $DOCS_OUTPUT_DIR"
node /tmp/validate-markdown.js "$DOCS_OUTPUT_DIR" 2>&1 || true
echo "::endgroup::"
# Validate OSS Tenant Structure outputs
echo "::group::Validate $REFERENCE_OUTPUT_PATH"
node /tmp/validate-markdown.js "$REFERENCE_OUTPUT_PATH" 2>&1 || true
echo "::endgroup::"
echo "::group::Validate $GETTING_STARTED_OUTPUT_PATH"
node /tmp/validate-markdown.js "$GETTING_STARTED_OUTPUT_PATH" 2>&1 || true
echo "::endgroup::"
echo "::group::Validate $DEVELOPMENT_OUTPUT_PATH"
node /tmp/validate-markdown.js "$DEVELOPMENT_OUTPUT_PATH" 2>&1 || true
echo "::endgroup::"
- name: Report stage 4 progress
if: always() && env.HUB_BASE_URL != ''
continue-on-error: true
env:
WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }}
WORKFLOW_RUN_ID: ${{ github.run_id }}
WORKFLOW_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }}
STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }}
STAGE2_STATUS: ${{ steps.stage2.outputs.stage2_status || steps.stage2_alt.outputs.stage2_status || 'skipped' }}
STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }}
STAGE3_STATUS: ${{ steps.stage3.outputs.stage3_status || 'skipped' }}
STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || 0 }}
STAGE4_STATUS: ${{ steps.stage4.outputs.stage4_status || 'skipped' }}
STAGE4_FILES: ${{ steps.stage4.outputs.stage4_files || 0 }}
PR_URL: ${{ steps.create-initial-pr.outputs.pull-request-url }}
PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }}
run: |
source /tmp/workflow-helpers.sh
CALLBACK_URL="${HUB_BASE_URL}/api/code-documentation/webhook"
echo "Reporting stage 4 (repository docs): $STAGE4_STATUS, $STAGE4_FILES files"
report_stage_progress "$CALLBACK_URL" "$WEBHOOK_SECRET" "$RUN_ID" "$REPO_ID" \
"$WORKFLOW_RUN_ID" "$WORKFLOW_URL" "creating-pr" \
"$STAGE1_STATUS" "$STAGE1_FILES" "$STAGE2_STATUS" "$STAGE2_FILES" "$STAGE3_STATUS" "$STAGE3_FILES" \
"$STAGE4_STATUS" "$STAGE4_FILES" "$PR_URL" "$PR_NUMBER"
# =========================================================================
# FINALIZE THE PULL REQUEST
# =========================================================================
- name: Clean up temporary files
run: |
source /tmp/workflow-helpers.sh
# Remove stats files
cleanup_path ".doc-stage1-stats.json"
cleanup_path ".doc-stage3-stats.json"
# Remove run status file (created for the initial PR).
# Legacy name kept so branches started before the rename still clean up.
cleanup_path ".flamingo-ai-technical-writer-status.md"
cleanup_path ".doc-pipeline-status.md"
# Note: .code-documentation-source-files.txt is now in /tmp/ (auto-cleanup)
# Remove npm artifacts (installed for scripts)
cleanup_path "node_modules"
cleanup_path "package.json"
cleanup_path "package-lock.json"
# NOTE: /tmp/workflow-helpers.sh is removed by "Report run result"
- name: Stage remaining docs and detect changes
id: stage-docs
run: |
source /tmp/workflow-helpers.sh
echo "Docs root: $DOCS_OUTPUT_PATH"
echo "Stage 2: $REFERENCE_OUTPUT_PATH (reference), $DIAGRAMS_OUTPUT_PATH (diagrams)"
echo "Stage 3: $GETTING_STARTED_OUTPUT_PATH (getting started), $DEVELOPMENT_OUTPUT_PATH (development)"
# Count untracked/modified files before staging
BEFORE_COUNT=$(git status --porcelain | wc -l)
echo "Changed files in the working tree: $BEFORE_COUNT"
# Stage ALL .md and .mmd files anywhere in the repo (for inline docs generated next to source files)
# This catches Stage 1 inline docs (hidden: .FileName.md), Stage 2 reference/diagrams, Stage 3 tutorials, and Stage 4 repo docs
# Find all .md and .mmd files recursively, including hidden files (.*.md)
# Includes README.md, CONTRIBUTING.md, LICENSE.md, SECURITY.md from Stage 4
# Includes .mmd Mermaid diagram files from Stage 2 (CodeWiki/Claude architecture)
find . \( -name "*.md" -o -name ".*.md" -o -name "*.mmd" \) -type f \
-not -path "./node_modules/*" \
-not -path "./.git/*" \
-not -name "CHANGELOG.md" \
-exec git add -f {} \; 2>/dev/null || true
echo "::group::Markdown and Mermaid files in the checkout (first 100)"
find . \( -name "*.md" -o -name ".*.md" -o -name "*.mmd" \) -type f \
-not -path "./node_modules/*" \
-not -path "./.git/*" \
-not -name "CHANGELOG.md" | head -100
echo "::endgroup::"
# Count staged files
STAGED_COUNT=$(git diff --cached --name-only | wc -l)
echo "Staged files: $STAGED_COUNT"
set_output "staged_count" "$STAGED_COUNT"
echo "::group::Staged files (first 50)"
git diff --cached --name-only | head -50
echo "::endgroup::"
# "Changes" means the BRANCH differs from the documented source head, not
# that this final sweep found something left to stage: every stage commits
# its own output as it goes, so a run whose stages all committed (inline
# docs, README) left nothing here and was reported as no_changes while its
# pull request held twenty files (openframe-saas-mobile run 35302244804).
# The status file is the run's own bookkeeping, never a documentation change.
if [ "$STAGED_COUNT" -eq "0" ] && git diff --quiet "$SOURCE_HEAD_SHA" HEAD -- . ':!.flamingo-ai-technical-writer-status.md'; then
echo "::notice title=No documentation changes::Nothing left to stage, and the branch holds no documentation change against the source head."
set_output "has_changes" "false"
else
set_output "has_changes" "true"
fi
- name: Mark pull request complete
if: steps.create-initial-pr.outputs.pull-request-number
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number }}
STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }}
STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }}
STAGE2_STATUS: ${{ steps.stage2.outputs.stage2_status || steps.stage2_alt.outputs.stage2_status || 'skipped' }}
STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }}
STAGE3_STATUS: ${{ steps.stage3.outputs.stage3_status || 'skipped' }}
STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || 0 }}
STAGE4_STATUS: ${{ steps.stage4.outputs.stage4_status || 'skipped' }}
STAGE4_FILES: ${{ steps.stage4.outputs.stage4_files || 0 }}
RUN_ID_VAR: ${{ env.RUN_ID }}
REPO_NAME: ${{ github.repository }}
run: |
source /tmp/workflow-helpers.sh
# Remove "in-progress" label
gh pr edit "$PR_NUMBER" --remove-label "in-progress" || true
# Update title to remove [IN PROGRESS]
gh pr edit "$PR_NUMBER" --title "🦩 Flamingo Code Documentation"
# Update body with final results
PR_BODY="## 🦩 Flamingo Code Documentation: Complete
**Run ID:** \`$RUN_ID_VAR\`
### Stage 1: Inline Documentation
- Status: $STAGE1_STATUS
- Files generated: $STAGE1_FILES
- Generated .md files next to source classes explaining their purpose
### Stage 2: Architecture Analysis
- Status: $STAGE2_STATUS
- Files generated: $STAGE2_FILES
- Architecture overview and module documentation
### Stage 3: AI Tutorial Generator
- Status: $STAGE3_STATUS
- Files generated: $STAGE3_FILES
- Getting started guides and how-to tutorials
### Stage 4: Repository Documentation
- Status: $STAGE4_STATUS
- Files generated: $STAGE4_FILES
- README.md, CONTRIBUTING.md, LICENSE.md, SECURITY.md
---
**Review checklist:**
- [ ] Check generated inline docs for accuracy
- [ ] Review architecture documentation
- [ ] Test code examples in tutorials
- [ ] Review README.md and CONTRIBUTING.md updates
---
🦩 Generated by [Flamingo Code Documentation](https://flamingo.run)"
gh pr edit "$PR_NUMBER" --body "$PR_BODY"
echo "Pull request #$PR_NUMBER marked complete"
# =========================================================================
# REPORT RUN RESULT: the terminal callback. Never reports "running".
# =========================================================================
- name: Report run result
if: always() && env.HUB_BASE_URL != ''
continue-on-error: true # Don't fail the workflow if callback fails
env:
# SECURITY: Pass secret per-step with inline masking
WEBHOOK_SECRET: ${{ secrets.FLAMINGO_HUB_SECRET }}
WORKFLOW_RUN_ID: ${{ github.run_id }}
WORKFLOW_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
HAS_CHANGES: ${{ steps.stage-docs.outputs.has_changes }}
JOB_STATUS: ${{ job.status }}
PR_URL: ${{ steps.create-initial-pr.outputs.pull-request-url }}
PR_NUMBER: ${{ steps.create-initial-pr.outputs.pull-request-number || 'null' }}
SAFE_RUN_ID: ${{ steps.branch-name-early.outputs.safe_run_id }}
# Single source of truth for the branch name (was rebuilt by hand below,
# which silently drifted from "Create docs branch" on every rename)
BRANCH_NAME: ${{ steps.create-pr-branch.outputs.branch_name }}
STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }}
STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || 0 }}
# Stage 2: Check both CodeWiki and Claude alternative, mark as failed if step failed
STAGE2_STATUS: ${{ steps.stage2.outcome == 'failure' && 'failed' || steps.stage2_alt.outcome == 'failure' && 'failed' || steps.stage2.outputs.stage2_status || steps.stage2_alt.outputs.stage2_status || 'skipped' }}
STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || 0 }}
STAGE3_STATUS: ${{ steps.stage3.outputs.stage3_status || 'skipped' }}
STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || 0 }}
STAGE4_STATUS: ${{ steps.stage4.outputs.stage4_status || 'skipped' }}
STAGE4_FILES: ${{ steps.stage4.outputs.stage4_files || 0 }}
# Track if critical steps failed (continue-on-error: false steps)
STAGE2_OUTCOME: ${{ steps.stage2.outcome || 'skipped' }}
CODEWIKI_INSTALL_OUTCOME: ${{ steps.codewiki_install.outcome || 'skipped' }}
CODEWIKI_CONFIG_OUTCOME: ${{ steps.codewiki_config.outcome || 'skipped' }}
run: |
# Bootstrap-failure fallback (shared failure-net standard with the
# code-review workflow): if workflow-helpers.sh never downloaded, no
# helper exists to report the failure — a minimal guarded curl posts
# it so the hub's run row fails NOW instead of waiting for the reaper.
if [ ! -f /tmp/workflow-helpers.sh ]; then
echo "::error title=Script bootstrap failed::workflow-helpers.sh was never downloaded from the hub; reporting the failure with a minimal callback."
# The bearer goes through a 0600 config file, never argv — see
# curlAuthPreamble in lib/config/workflow-scripts-bootstrap.ts.
CURL_CFG=$(mktemp) && chmod 600 "$CURL_CFG"
trap 'rm -f "$CURL_CFG"' EXIT
printf 'header = "Authorization: Bearer %s"\n' "$WEBHOOK_SECRET" > "$CURL_CFG"
curl -sS --max-time 30 -K "$CURL_CFG" -X POST "${HUB_BASE_URL}/api/code-documentation/webhook" \
-H "Content-Type: application/json" \
-d "{\"run_id\":\"$RUN_ID\",\"repo_id\":\"$REPO_ID\",\"status\":\"failure\",\"workflow_run_id\":$WORKFLOW_RUN_ID,\"workflow_url\":\"$WORKFLOW_URL\",\"error\":\"Script bootstrap failed: workflow-helpers.sh never downloaded from the hub.\"}" || true
exit 1
fi
source /tmp/workflow-helpers.sh
CALLBACK_URL="${HUB_BASE_URL}/api/code-documentation/webhook"
echo "::group::Inputs to the final status"
echo "JOB_STATUS=$JOB_STATUS"
echo "HAS_CHANGES=$HAS_CHANGES"
echo "STAGE2_OUTCOME=$STAGE2_OUTCOME"
echo "CODEWIKI_INSTALL_OUTCOME=$CODEWIKI_INSTALL_OUTCOME"
echo "CODEWIKI_CONFIG_OUTCOME=$CODEWIKI_CONFIG_OUTCOME"
echo "::endgroup::"
# CRITICAL: Determine final status - NEVER return "running"
# Default to failure, only set success if everything checks out
STATUS="failure"
# Check for cancelled job first
if [ "$JOB_STATUS" = "cancelled" ]; then
STATUS="cancelled"
REASON="the workflow run was cancelled"
# Check if critical stage 2 (CodeWiki) failed - this has continue-on-error: false
elif [ "$STAGE2_OUTCOME" = "failure" ]; then
STATUS="failure"
REASON="the CodeWiki stage failed"
# Check if CodeWiki installation failed
elif [ "$CODEWIKI_INSTALL_OUTCOME" = "failure" ]; then
STATUS="failure"
REASON="the CodeWiki installation failed"
# Check if CodeWiki configuration failed
elif [ "$CODEWIKI_CONFIG_OUTCOME" = "failure" ]; then
STATUS="failure"
REASON="the CodeWiki configuration failed"
# Check overall job status
elif [ "$JOB_STATUS" != "success" ]; then
STATUS="failure"
REASON="the job status is $JOB_STATUS"
# A requested stage that FAILED fails the run, whatever the others wrote:
# a run whose stages 3 and 4 wrote nothing used to report success because
# stage 2 had committed reference docs.
elif FAILED_STAGES=$(for n in 1 2 3 4; do v="STAGE${n}_STATUS"; [ "${!v}" = "failed" ] && printf 'stage %s, ' "$n"; done) && [ -n "$FAILED_STAGES" ]; then
STATUS="failure"
REASON="${FAILED_STAGES%, } failed"
# Check if we have any documentation changes
elif [ "$HAS_CHANGES" != "true" ]; then
STATUS="no_changes"
REASON="no documentation changed"
else
STATUS="success"
REASON="documentation generated"
fi
# SAFETY CHECK: Ensure status is NEVER "running"
if [ "$STATUS" = "running" ] || [ -z "$STATUS" ]; then
echo "::warning::Computed status '$STATUS' is not terminal; reporting failure instead."
STATUS="failure"
REASON="no terminal status could be determined"
fi
case "$STATUS" in
success) echo "Final status: success ($REASON)" ;;
no_changes) echo "::notice title=Code Documentation: no changes::Final status: no_changes ($REASON)." ;;
cancelled) echo "::warning title=Code Documentation: cancelled::Final status: cancelled ($REASON)." ;;
*) echo "::error title=Code Documentation: $STATUS::Final status: $STATUS ($REASON)." ;;
esac
# Branch actually created by "Create docs branch". Falls back to the same
# formula only when that step never ran (this step is `if: always()`).
SAFE_BRANCH="${BRANCH_NAME:-${DOCS_BRANCH_PREFIX}$SAFE_RUN_ID}"
# Send final webhook using helper function
report_final_status "$CALLBACK_URL" "$WEBHOOK_SECRET" "$RUN_ID" "$REPO_ID" \
"$WORKFLOW_RUN_ID" "$WORKFLOW_URL" "$STATUS" "$PR_URL" "$PR_NUMBER" "$SAFE_BRANCH" \
"$STAGE1_STATUS" "$STAGE1_FILES" "$STAGE2_STATUS" "$STAGE2_FILES" "$STAGE3_STATUS" "$STAGE3_FILES" \
"$STAGE4_STATUS" "$STAGE4_FILES"
# Final cleanup: remove workflow helpers file
cleanup_path "/tmp/workflow-helpers.sh"
# The run's job summary (the Actions run page). Values reach the script
# through env, never inline expressions.
- name: Write job summary
if: always()
env:
STAGE1_STATUS: ${{ steps.stage1.outputs.stage1_status || 'skipped' }}
STAGE1_FILES: ${{ steps.stage1.outputs.stage1_files || '0' }}
STAGE2_STATUS: ${{ steps.stage2.outputs.stage2_status || steps.stage2_alt.outputs.stage2_status || 'skipped' }}
STAGE2_FILES: ${{ steps.stage2.outputs.stage2_files || steps.stage2_alt.outputs.stage2_files || '0' }}
STAGE2_ENGINE: ${{ steps.detect_language.outputs.codewiki_supported == 'true' && 'CodeWiki' || steps.detect_language.outputs.codewiki_supported == 'false' && 'Claude' || 'engine not chosen' }}
STAGE3_STATUS: ${{ steps.stage3.outputs.stage3_status || 'skipped' }}
STAGE3_FILES: ${{ steps.stage3.outputs.stage3_files || '0' }}
STAGE4_STATUS: ${{ steps.stage4.outputs.stage4_status || 'skipped' }}
STAGE4_FILES: ${{ steps.stage4.outputs.stage4_files || '0' }}
GRAPH_OUTCOME: ${{ steps.graph.outcome || 'skipped' }}
PRIMARY_LANGUAGE: ${{ steps.detect_language.outputs.primary_language }}
PR_URL: ${{ steps.create-initial-pr.outputs.pull-request-url }}
run: |
{
echo "## 🦩 Flamingo Code Documentation"
echo ""
echo "Run \`$RUN_ID\` · source branch \`$SOURCE_BRANCH\` · primary language ${PRIMARY_LANGUAGE:-unknown}"
echo ""
echo "| Stage | Status | Files |"
echo "|-------|--------|-------|"
echo "| 0 · Code graph | $GRAPH_OUTCOME | – |"
echo "| 1 · Inline docs | $STAGE1_STATUS | $STAGE1_FILES |"
echo "| 2 · Reference docs ($STAGE2_ENGINE) | $STAGE2_STATUS | $STAGE2_FILES |"
echo "| 3 · Tutorials | $STAGE3_STATUS | $STAGE3_FILES |"
echo "| 4 · Repository docs | $STAGE4_STATUS | $STAGE4_FILES |"
echo ""
if [ -n "$PR_URL" ]; then
echo "**Pull request:** $PR_URL"
else
echo "No pull request was opened."
fi
} >> "$GITHUB_STEP_SUMMARY"