name: CI on: push: branches: ["**"] pull_request: branches: ["**"] jobs: guardrails: runs-on: ubuntu-latest steps: - uses: actions/checkout@v4 # Defense in depth: fail if secrets or heavy/generated dirs were ever committed. - name: Block secrets & node_modules run: | if git ls-files | grep -E '(^|/)\.env($|\.)|(^|/)secrets/|\.pem$|\.key$|\.p12$|\.pfx$|(^|/)id_rsa$|(^|/)id_ed25519$'; then echo "::error::Secret-like files are tracked — remove them and rotate any exposed credential."; exit 1 fi if git ls-files | grep -E '(^|/)node_modules/'; then echo "::error::node_modules is tracked — it must be gitignored."; exit 1 fi - name: Required standards files present run: | for f in CLAUDE.md PROJECT_STANDARD.md README.md; do test -f "$f" || { echo "::error::Missing required file: $f"; exit 1; } done tests: runs-on: ubuntu-latest steps: - uses: actions/checkout@v4 - uses: actions/setup-python@v5 with: python-version: "3.13" - name: Install dependencies run: | python -m pip install --upgrade pip pip install -r orchestrator/requirements.txt -r agent/requirements.txt # Hermetic: in-memory store, planner forced offline by the tests — no model key needed. - name: Manager contract test env: { REDIS_FAKE: "1" } run: python scripts/test-runtime-contract.py - name: Workflow mechanism smoke test env: { REDIS_FAKE: "1" } run: python scripts/test-merge-smoke.py - name: End-to-end workflow test env: { REDIS_FAKE: "1" } run: python scripts/test-workflow-e2e.py - name: Manager event-contract test env: { REDIS_FAKE: "1" } run: python scripts/test-contract-events.py - name: Benchmark metric formulas (v2.1) env: { REDIS_FAKE: "1" } run: python scripts/test-benchmark-metrics.py - name: Benchmark collector env: { REDIS_FAKE: "1" } run: python scripts/test-benchmark-collector.py - name: Baseline comparison env: { REDIS_FAKE: "1" } run: python scripts/test-baseline-comparison.py - name: Code sandbox (in-pod test runner) run: python scripts/test-sandbox.py - name: Quality instrumentation (Group B) env: { REDIS_FAKE: "1" } run: python scripts/test-quality.py - name: Decision-engine pheromone library (τ) env: { REDIS_FAKE: "1" } run: python scripts/test-decision-engine.py - name: Dispatch scoring formulas env: { REDIS_FAKE: "1" } run: python scripts/test-dispatch-score.py # --- decentralized swarm flow (the only flow; primitives are unconditional) --- - name: Swarm seeder (#6) run: python scripts/test-swarm-seed.py - name: Swarm self-selection dispatch env: { REDIS_FAKE: "1" } run: python scripts/test-swarm-dispatch.py - name: Swarm autonomous task generation (#7) env: { REDIS_FAKE: "1", AGENT_PROPOSAL_BUDGET: "3" } run: python scripts/test-swarm-autonomous.py - name: Swarm task competition (#8) env: { REDIS_FAKE: "1" } run: python scripts/test-swarm-competition.py - name: Swarm cross-review (#11) env: { REDIS_FAKE: "1" } run: python scripts/test-swarm-cross-review.py - name: Swarm convergence (#12) env: { REDIS_FAKE: "1" } run: python scripts/test-swarm-convergence.py - name: Swarm health guard env: { REDIS_FAKE: "1" } run: python scripts/test-swarm-guard.py # Pure-module unit tests for the swarm primitives (formulas/policies, infra-free). - name: Swarm primitive modules (unit) env: { REDIS_FAKE: "1" } run: | python scripts/test-autonomous-tasks.py python scripts/test-task-competition.py python scripts/test-cross-review.py python scripts/test-convergence.py # --- benchmark baseline runners + suite smoke (Group C, #21) --- # HEICODE_SANDBOX_ISOLATED: the runners grade generated code in the fail-closed sandbox; the # CI runner is ephemeral/isolated, so confirm isolation here (see security-boundary §8.1). - name: Benchmark runners (Group C) env: { HEICODE_SANDBOX_ISOLATED: "1" } run: python scripts/test-benchmark-runners.py # Offline = pipeline validation only (deterministic, no real G_E). Guards against regressions. - name: Benchmark suite smoke (offline) env: { HEICODE_SANDBOX_ISOLATED: "1" } run: python scripts/run-benchmark-suite.py --taskset coding-set-1