mirror of
https://github.com/NousResearch/hermes-agent.git
synced 2026-09-28 06:45:17 +08:00
tests/ci/test_stable_release_graph.py requires every reusable-workflow job in all-checks-pass; keep it there and let a red shard show without failing the gate. Log a resource trail during the run: every shard's runner died ~2 minutes into seeding on the first CI run with only 'The runner has received a shutdown signal'.
459 lines
20 KiB
YAML
459 lines
20 KiB
YAML
name: CI
|
|
|
|
# Orchestrator workflow. Runs ``detect-changes`` once, then conditionally
|
|
# calls the sub-workflows that a PR can actually affect. A final
|
|
# ``all-checks-pass`` gate job aggregates results so branch protection only
|
|
# needs to require a single check.
|
|
#
|
|
# Sub-workflows are triggered via ``workflow_call`` and keep their own job
|
|
# definitions, matrices, and concurrency settings. They no longer have
|
|
# ``push:`` / ``pull_request:`` triggers of their own — everything flows
|
|
# through this file.
|
|
#
|
|
# SECURITY: this workflow runs PR-controlled actions, workflows, and code.
|
|
# Do not add ``secrets: inherit`` or GitHub App credentials here. Trusted
|
|
# main-only automation uses protected environments in its own workflows.
|
|
|
|
on:
|
|
workflow_dispatch:
|
|
inputs:
|
|
release:
|
|
description: 'Stable-release candidate run: force every applicability lane and make the aggregate gate strict (skipped required lanes fail).'
|
|
required: false
|
|
type: boolean
|
|
default: false
|
|
workflow_call:
|
|
inputs:
|
|
release:
|
|
description: 'Stable-release candidate run: force every applicability lane and make the aggregate gate strict (skipped required lanes fail).'
|
|
required: false
|
|
type: boolean
|
|
default: false
|
|
pull_request:
|
|
push:
|
|
branches: [main]
|
|
|
|
permissions:
|
|
contents: read
|
|
pull-requests: write # needed by lint (PR comment) + supply-chain review_status
|
|
|
|
# cancel-in-progress only ever applies to PR events. A push, a dispatch, and
|
|
# above all a stable-release workflow_call run are never cancelled by a later
|
|
# commit or a rerun of the same branch — the caller's own concurrency uses
|
|
# github.run_id, so even a parent rerun cannot kill this child mid-flight.
|
|
# Release (workflow_call with inputs.release) never cancels and never gets
|
|
# cancelled: its group is github.run_id. PRs collapse per-PR; pushes use the
|
|
# ref as before.
|
|
concurrency:
|
|
group: ci-${{ inputs.release == true && github.run_id || (github.event_name == 'pull_request' && github.event.pull_request.number || github.ref) }}
|
|
cancel-in-progress: ${{ inputs.release != true && github.event_name == 'pull_request' }}
|
|
|
|
jobs:
|
|
# ─────────────────────────────────────────────────────────────────────
|
|
# detect: run the classifier once. Every downstream job reads its outputs
|
|
# to decide whether to run. On push/dispatch the classifier fails open
|
|
# (all lanes true) so post-merge validation is never weakened.
|
|
#
|
|
# A release run (inputs.release) additionally forces every lane true via
|
|
# the `release-forced-*` outputs: a stable candidate must run the FULL
|
|
# pipeline, not the lanes its diff would touch — the diff of a release
|
|
# tag is not a meaningful applicability signal.
|
|
# ─────────────────────────────────────────────────────────────────────
|
|
detect:
|
|
name: Detect affected areas
|
|
runs-on: ubuntu-latest
|
|
timeout-minutes: 1
|
|
outputs:
|
|
python: ${{ steps.gate-lanes.outputs.python }}
|
|
python_prod: ${{ steps.gate-lanes.outputs.python_prod }}
|
|
frontend: ${{ steps.gate-lanes.outputs.frontend }}
|
|
site: ${{ steps.gate-lanes.outputs.site }}
|
|
scan: ${{ steps.gate-lanes.outputs.scan }}
|
|
deps: ${{ steps.gate-lanes.outputs.deps }}
|
|
uv_lock: ${{ steps.gate-lanes.outputs.uv_lock }}
|
|
npm_lock: ${{ steps.gate-lanes.outputs.npm_lock }}
|
|
bootstrap: ${{ steps.gate-lanes.outputs.bootstrap }}
|
|
desktop_updater: ${{ steps.gate-lanes.outputs.desktop_updater }}
|
|
rust: ${{ steps.gate-lanes.outputs.rust }}
|
|
docker_meta: ${{ steps.gate-lanes.outputs.docker_meta }}
|
|
mcp_catalog: ${{ steps.gate-lanes.outputs.mcp_catalog }}
|
|
ci_review: ${{ steps.classify.outputs.ci_review }}
|
|
ci_review_files: ${{ steps.classify.outputs.ci_review_files }}
|
|
event_name: ${{ github.event_name }}
|
|
steps:
|
|
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
|
- name: Detect affected areas
|
|
id: classify
|
|
uses: ./.github/actions/detect-changes
|
|
with:
|
|
github-token: ${{ github.token }}
|
|
- name: Force all lanes on release
|
|
# Overwrite the classifier outputs with 'true' when inputs.release.
|
|
# ci_review stays raw: it only ever feeds the PR review-label gate.
|
|
id: gate-lanes
|
|
env:
|
|
RELEASE: ${{ inputs.release }}
|
|
CLASSIFIED: ${{ toJSON(steps.classify.outputs) }}
|
|
run: |
|
|
python3 - <<'PY'
|
|
import json, os
|
|
values = json.loads(os.environ['CLASSIFIED'])
|
|
with open(os.environ['GITHUB_OUTPUT'], 'a', encoding='utf-8') as output:
|
|
for lane, value in values.items():
|
|
if os.environ['RELEASE'] == 'true' and value in ('true', 'false'):
|
|
value = 'true'
|
|
if '\n' in value or '\r' in value:
|
|
continue
|
|
output.write(f'{lane}={value}\n')
|
|
PY
|
|
|
|
# ─────────────────────────────────────────────────────────────────────
|
|
# Lane-gated sub-workflows. Each runs in parallel after detect finishes.
|
|
# Skipped workflows (if condition is false) don't spin up runners.
|
|
# ─────────────────────────────────────────────────────────────────────
|
|
tests:
|
|
name: Python tests
|
|
needs: detect
|
|
if: needs.detect.outputs.python == 'true'
|
|
uses: ./.github/workflows/tests.yml
|
|
|
|
# macOS + Windows lanes. The main `tests` lane above is Linux-only, and
|
|
# the OS-marked tests it collects are skipped there by design (see the
|
|
# `_OS_MARKS` comment in tests/conftest.py) — this is where they run.
|
|
# Same `python` lane gate: if no Python changed, neither runs.
|
|
tests-os:
|
|
name: OS-specific tests
|
|
needs: detect
|
|
if: needs.detect.outputs.python == 'true'
|
|
uses: ./.github/workflows/tests-os.yml
|
|
with:
|
|
# The Windows lane spawns the real desktop-update hand-off script
|
|
# (tests/test_desktop_update_windows_*.py) only when that surface
|
|
# changed; unit-level platforms("windows") tests always run.
|
|
desktop_updater: ${{ needs.detect.outputs.desktop_updater == 'true' }}
|
|
|
|
lint:
|
|
name: Python lints
|
|
needs: detect
|
|
if: needs.detect.outputs.python == 'true'
|
|
uses: ./.github/workflows/lint.yml
|
|
with:
|
|
event_name: ${{ needs.detect.outputs.event_name }}
|
|
|
|
js-tests:
|
|
name: JS & TS checks
|
|
needs: detect
|
|
if: needs.detect.outputs.frontend == 'true'
|
|
uses: ./.github/workflows/js-tests.yml
|
|
|
|
rust-tests:
|
|
name: Rust tests
|
|
needs: detect
|
|
# Only for PRs that touch a Rust crate. `.rs` is under apps/, so these
|
|
# changes used to run the TypeScript matrix and nothing that compiles them.
|
|
if: needs.detect.outputs.rust == 'true'
|
|
uses: ./.github/workflows/rust-tests.yml
|
|
|
|
bootstrap-installer:
|
|
name: Bootstrap installer
|
|
needs: detect
|
|
# The bootstrap-installer path: install.sh, the pin fragments it embeds,
|
|
# and the version stamp it ships. The PowerShell installer's native tests
|
|
# are platforms("windows") pytest files and run in tests-os.
|
|
if: needs.detect.outputs.bootstrap == 'true'
|
|
uses: ./.github/workflows/bootstrap-installer.yml
|
|
|
|
e2e-desktop:
|
|
name: Desktop E2E
|
|
needs: detect
|
|
# python_prod (not python): the Playwright suite exercises the built app
|
|
# + `hermes serve` backend, which never import anything under tests/.
|
|
# Tests-only PRs (~17% of commits) skip this 5-minute job — the longest
|
|
# single job in the workflow — while still running the full pytest lanes.
|
|
#
|
|
# Re-disabled (Sep 2026): the Sep 1 re-enable is still incredibly flaky.
|
|
# Keep this a bare `if: false`. The earlier
|
|
# `${{ false && (... || ...) }}` form on this reusable-workflow job made
|
|
# GitHub's workflow parser fail at startup ("An unexpected error has
|
|
# occurred") — every ci.yaml run repo-wide dispatched 0 jobs from
|
|
# 24f5a60ed1 until this line changed. To re-enable, restore:
|
|
# if: ${{ needs.detect.outputs.python_prod == 'true' || needs.detect.outputs.frontend == 'true' }}
|
|
if: false
|
|
uses: ./.github/workflows/e2e-desktop.yml
|
|
|
|
e2e-desktop-core:
|
|
name: Desktop core E2E
|
|
needs: detect
|
|
# The deterministic core suite (apps/desktop/e2e/core): transcript
|
|
# integrity, boot/respawn/orphans, clarify+approval. Required; retries 0.
|
|
if: ${{ needs.detect.outputs.python_prod == 'true' || needs.detect.outputs.frontend == 'true' }}
|
|
uses: ./.github/workflows/e2e-desktop-core.yml
|
|
|
|
e2e-desktop-update:
|
|
name: Desktop update E2E
|
|
needs: detect
|
|
# apps/desktop/e2e/update: a real install.sh install + packaged Desktop,
|
|
# first run, "Update now" hand-off/relaunch, concurrent-update refusal.
|
|
# Advisory until it has proven stable on CI: its shards run with continue-on-error, so a
|
|
# red shard shows on the PR but does not fail all-checks-pass.
|
|
if: ${{ needs.detect.outputs.python_prod == 'true' || needs.detect.outputs.frontend == 'true' }}
|
|
uses: ./.github/workflows/e2e-desktop-update.yml
|
|
|
|
docs-site:
|
|
name: Docs Site
|
|
needs: detect
|
|
if: needs.detect.outputs.site == 'true'
|
|
uses: ./.github/workflows/docs-site-checks.yml
|
|
|
|
history-check:
|
|
name: Deny unrelated histories
|
|
needs: detect
|
|
if: needs.detect.outputs.event_name == 'pull_request'
|
|
uses: ./.github/workflows/history-check.yml
|
|
|
|
contributor-check:
|
|
name: Check contributors
|
|
needs: detect
|
|
if: needs.detect.outputs.python == 'true'
|
|
uses: ./.github/workflows/contributor-check.yml
|
|
|
|
uv-lockfile:
|
|
name: Check uv.lock
|
|
needs: detect
|
|
# Gated: `uv lock --check` re-resolves the whole dependency graph against
|
|
# PyPI, so on every PR it spent a network round-trip — and, on a registry
|
|
# blip, a blocking red X — for diffs that cannot desync the lockfile
|
|
# (docs, frontend, prose). Only pyproject.toml / uv.lock can. A
|
|
# `.github/` change still forces it on via the classifier's fail-open.
|
|
if: needs.detect.outputs.uv_lock == 'true'
|
|
uses: ./.github/workflows/uv-lockfile-check.yml
|
|
|
|
infographic-check:
|
|
name: Check no committed infographics
|
|
needs: detect
|
|
uses: ./.github/workflows/infographic-check.yml
|
|
|
|
profile-artifact-check:
|
|
name: Profile artifact check
|
|
needs: detect
|
|
uses: ./.github/workflows/profile-artifact-check.yml
|
|
|
|
icons-freshness-check:
|
|
name: Icon assets freshness
|
|
needs: detect
|
|
uses: ./.github/workflows/icons-freshness-check.yml
|
|
|
|
case-collision-check:
|
|
name: Check no case-colliding filenames
|
|
needs: detect
|
|
uses: ./.github/workflows/case-collision-check.yml
|
|
|
|
lazy-deps-guard:
|
|
name: No production imports of the tools.lazy_deps stub
|
|
needs: detect
|
|
uses: ./.github/workflows/lazy-deps-guard.yml
|
|
|
|
lockfile-diff:
|
|
name: package-lock.json diff
|
|
needs: detect
|
|
if: needs.detect.outputs.event_name == 'pull_request' && needs.detect.outputs.npm_lock == 'true'
|
|
uses: ./.github/workflows/lockfile-diff.yml
|
|
|
|
docker-lint:
|
|
name: Lint Docker scripts
|
|
needs: detect
|
|
if: needs.detect.outputs.docker_meta == 'true'
|
|
uses: ./.github/workflows/docker-lint.yml
|
|
|
|
supply-chain:
|
|
name: Supply-chain scan
|
|
needs: detect
|
|
if: needs.detect.outputs.event_name == 'pull_request' && (needs.detect.outputs.scan == 'true' || needs.detect.outputs.deps == 'true')
|
|
uses: ./.github/workflows/supply-chain-audit.yml
|
|
with:
|
|
event_name: ${{ needs.detect.outputs.event_name }}
|
|
scan: ${{ needs.detect.outputs.scan == 'true' }}
|
|
deps: ${{ needs.detect.outputs.deps == 'true' }}
|
|
|
|
review-labels:
|
|
name: Review label gate
|
|
needs: [detect, supply-chain]
|
|
if: always() && needs.detect.outputs.event_name == 'pull_request' && (needs.detect.outputs.ci_review == 'true' || needs.detect.outputs.mcp_catalog == 'true' || needs.supply-chain.outputs.critical_findings == 'true')
|
|
uses: ./.github/workflows/review-labels.yml
|
|
with:
|
|
ci_review: ${{ needs.detect.outputs.ci_review == 'true' }}
|
|
ci_review_files: ${{ needs.detect.outputs.ci_review_files }}
|
|
mcp_catalog: ${{ needs.detect.outputs.mcp_catalog == 'true' }}
|
|
supply_chain: ${{ needs.supply-chain.outputs.critical_findings == 'true' }}
|
|
|
|
# ─────────────────────────────────────────────────────────────────────
|
|
# Gate: runs after everything. ``if: always()`` ensures it reports a
|
|
# status even when some deps were skipped.
|
|
#
|
|
# Non-release (PR/push): failure fails, skipped counts as success.
|
|
# Release (inputs.release): strict — every required job must be
|
|
# `success`. A skipped required lane fails the gate; only the PR-only
|
|
# jobs (history-check, lockfile-diff, supply-chain, review-labels) and
|
|
# the deferred Desktop E2E (e2e-desktop) may skip. The OSV scan's
|
|
# findings are advisory, but its execution is required.
|
|
#
|
|
# Branch protection should require ONLY this check.
|
|
#
|
|
# Outputs ``needs-json`` — a compact ``{job_name: result}`` dict — so
|
|
# the live comment poller can list failed jobs in the PR comment.
|
|
# ─────────────────────────────────────────────────────────────────────
|
|
all-checks-pass:
|
|
name: All required checks pass
|
|
needs:
|
|
- detect
|
|
- tests
|
|
- tests-os
|
|
- lint
|
|
- js-tests
|
|
- rust-tests
|
|
- bootstrap-installer
|
|
- e2e-desktop
|
|
- e2e-desktop-core
|
|
- e2e-desktop-update
|
|
- docs-site
|
|
- history-check
|
|
- contributor-check
|
|
- uv-lockfile
|
|
- infographic-check
|
|
- case-collision-check
|
|
- lazy-deps-guard
|
|
- lockfile-diff
|
|
- docker-lint
|
|
- profile-artifact-check
|
|
- icons-freshness-check
|
|
- supply-chain
|
|
- review-labels
|
|
# OSV runs weekly against main (osv-scanner.yml schedule), not per PR:
|
|
# every PR was reporting the same repo-wide baseline of pinned-dep CVEs
|
|
# in its review comment, and the SARIF upload tripped GitHub's
|
|
# per-installation API rate limit during merge trains.
|
|
# The image build runs in its own workflow (docker.yml) and reports
|
|
# its own check. It was never required here, because it is too slow
|
|
# to block a merge. A separate run also stops it from holding this
|
|
# run open. That is what blocked ``gh run rerun``.
|
|
if: always()
|
|
runs-on: ubuntu-latest
|
|
timeout-minutes: 10
|
|
outputs:
|
|
needs-json: ${{ steps.evaluate.outputs.needs-json }}
|
|
steps:
|
|
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
|
with:
|
|
ref: ${{ github.sha }}
|
|
- name: Evaluate job results
|
|
id: evaluate
|
|
# Shared with the stable orchestrator (scripts/ci/required_results.py
|
|
# is imported there). NEEDS is the toJSON(needs) context; RELEASE
|
|
# switches the gate into strict mode for release runs.
|
|
env:
|
|
NEEDS: ${{ toJSON(needs) }}
|
|
RELEASE: ${{ inputs.release }}
|
|
run: |
|
|
args=()
|
|
if [ "$RELEASE" = true ]; then args+=(--release); fi
|
|
printf '%s' "$NEEDS" | python3 scripts/ci/required_results.py "${args[@]}"
|
|
|
|
# ─────────────────────────────────────────────────────────────────────
|
|
# CI timing report: collect per-job/step durations from the GitHub API,
|
|
# cache them on main (as a baseline), and on PRs generate an HTML diff
|
|
# report with a gantt chart + per-step breakdown. The report is uploaded
|
|
# as an artifact and a markdown summary is written to $GITHUB_STEP_SUMMARY.
|
|
#
|
|
# The live comment poller dynamically fetches all review-status-* artifacts
|
|
# across the orchestrator and sub-workflow runs every cycle, so its link
|
|
# points straight at that report.
|
|
# ─────────────────────────────────────────────────────────────────────
|
|
ci-timings:
|
|
name: CI timing report
|
|
needs: [all-checks-pass]
|
|
if: always()
|
|
runs-on: ubuntu-latest
|
|
timeout-minutes: 10
|
|
steps:
|
|
- name: Checkout code
|
|
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
|
|
|
- name: Restore baseline cache (PR only)
|
|
if: github.event_name == 'pull_request'
|
|
uses: actions/cache/restore@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5
|
|
with:
|
|
path: ci-timings-baseline.json
|
|
# Prefix-match: exact key will never hit (run_id differs), so
|
|
# restore-keys finds the most recent baseline from main.
|
|
key: ci-timings-baseline-never-exact
|
|
restore-keys: |
|
|
ci-timings-baseline-
|
|
|
|
- name: Collect timings and generate report
|
|
env:
|
|
GITHUB_TOKEN: ${{ github.token }}
|
|
run: |
|
|
python3 scripts/ci/timings_report.py \
|
|
--baseline ci-timings-baseline.json \
|
|
--output ci-timings-report.html \
|
|
--json-out ci-timings.json \
|
|
--summary-out ci-timings-summary.md
|
|
|
|
- name: Upload HTML report
|
|
# Advisory report — artifact-service blips must not fail the job.
|
|
continue-on-error: true
|
|
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
|
id: ci-timings-html
|
|
with:
|
|
name: ci-timings-report
|
|
path: ci-timings-report.html
|
|
retention-days: 14
|
|
|
|
- name: Build linked review status
|
|
if: hashFiles('ci-timings.json') != ''
|
|
env:
|
|
CI_TIMINGS_REPORT_URL: ${{ steps.ci-timings-html.outputs.artifact-url }}
|
|
run: |
|
|
python3 scripts/ci/timings_report.py \
|
|
--from-json ci-timings.json \
|
|
--baseline ci-timings-baseline.json \
|
|
--review-status-out review-status.json \
|
|
--review-status-only
|
|
|
|
- name: Upload review status
|
|
if: hashFiles('review-status.json') != ''
|
|
continue-on-error: true
|
|
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
|
with:
|
|
name: review-status-ci-timings
|
|
path: review-status.json
|
|
retention-days: 14
|
|
|
|
- name: Output summary
|
|
env:
|
|
REPORT_URL: ${{ steps.ci-timings-html.outputs.artifact-url}}
|
|
run: |
|
|
{
|
|
echo "# CI Timing report"
|
|
echo "[View the full interactive report]($REPORT_URL)"
|
|
} >> "$GITHUB_STEP_SUMMARY"
|
|
cat ci-timings-summary.md >> "$GITHUB_STEP_SUMMARY"
|
|
|
|
- name: Save baseline cache (main only)
|
|
if: github.event_name == 'push' && github.ref == 'refs/heads/main'
|
|
run: |
|
|
# Degraded runs (API rate-limited) produce no ci-timings.json —
|
|
# skip rather than fail, and never cache an empty baseline.
|
|
if [ -f ci-timings.json ]; then
|
|
cp ci-timings.json ci-timings-baseline.json
|
|
else
|
|
echo "No timings JSON this run — skipping baseline update"
|
|
fi
|
|
|
|
- name: Upload baseline to cache (main only)
|
|
if: github.event_name == 'push' && github.ref == 'refs/heads/main' && hashFiles('ci-timings-baseline.json') != ''
|
|
uses: actions/cache/save@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5
|
|
with:
|
|
path: ci-timings-baseline.json
|
|
key: ci-timings-baseline-${{ github.run_id }}
|