Skip to content

System Tests

System Tests #161

Workflow file for this run

name: System Tests
on:
pull_request:
types: [opened, synchronize, reopened]
issue_comment:
types: [created]
workflow_dispatch:
inputs:
marks:
description: "pytest marks expression (e.g. 'build_docker', 'liveliness', 'takeoff_hover_land'). \
Use 'or' to combine marks: 'liveliness or takeoff_hover_land'. Leave blank to run all marks. \
Note: 'build_packages' is automatically prepended whenever any marks are specified, \
to ensure code is built before launch tests run."
default: "liveliness or takeoff_hover_land"
required: false
sim:
description: "Sim targets, comma-separated: isaacsim,msairsim. Default isaacsim; pass msairsim to opt in."
default: isaacsim
required: false
num_robots:
description: "Robot counts, comma-separated (e.g. 1,3)"
default: "1"
required: false
stress_iterations:
description: "Iterations per (sim, num_robots) config"
default: "1"
required: false
stable_duration:
description: "Seconds for test_stable polling window"
default: "120"
required: false
trajectory_types:
description: "Fixed trajectories, comma-separated (e.g. Circle or Circle,Figure8)"
default: "Circle,Figure8,Racetrack,Line"
required: false
takeoff_velocities:
description: "Takeoff velocities, comma-separated (e.g. 0.5 or 0.5,1)"
default: "0.5"
required: false
baseline_run_id:
description: "Run ID to use as baseline for metric comparison (blank = latest successful run on main)"
default: ""
required: false
jobs:
run-tests:
name: Run Tests
runs-on: [self-hosted, airstack-ephemeral]
# Triggers:
# - workflow_dispatch (manual)
# - PR opened, synchronized, or reopened from the same repo — same-repo guard
# prevents arbitrary code execution on the self-hosted runner from
# untrusted contributors.
# - PR comment starting with `/pytest` from a user with write access
# (OWNER/MEMBER/COLLABORATOR). issue_comment fires for both issues
# and PRs; `issue.pull_request` disambiguates. The author_association
# gate is what keeps random commenters from running code on the
# self-hosted runner.
if: |
github.event_name == 'workflow_dispatch' ||
(github.event_name == 'pull_request' &&
github.event.pull_request.head.repo.full_name == github.repository) ||
(github.event_name == 'issue_comment' &&
github.event.issue.pull_request != null &&
startsWith(github.event.comment.body, '/pytest') &&
contains(fromJSON('["OWNER","MEMBER","COLLABORATOR"]'), github.event.comment.author_association))
concurrency:
group: system-tests-${{ github.event.pull_request.number || github.event.issue.number || github.run_id }}
cancel-in-progress: true
timeout-minutes: 120
# Adding any `permissions:` entry disables GITHUB_TOKEN's defaults, so
# every scope used here has to be re-granted explicitly:
# checks:write — create/update the Check Run on the PR head
# contents:read — actions/checkout
# pull-requests:write — post the acknowledgment comment on the PR.
# Even though the endpoint is /issues/{n}/comments, comments on
# PRs are gated by the pull-requests permission, not issues — the
# `x-accepted-github-permissions` header lists both as alternatives
# but only pull-requests:write actually works for PR comments.
permissions:
checks: write
contents: read
pull-requests: write
# Mirror the registry password into env so step-level `if:` expressions
# can check whether registry-cache mode is available — `secrets.*` itself
# is not addressable from `if:` expressions.
env:
DOCKER_REGISTRY_PASSWORD: ${{ secrets.DOCKER_REGISTRY_PASSWORD }}
outputs:
tested_sha: ${{ steps.identity.outputs.tested_sha }}
pr_number: ${{ steps.identity.outputs.pr_number }}
check_run_id: ${{ steps.check_create.outputs.id }}
steps:
# Uses actions/github-script (Node, bundled with the runner) instead
# of `gh` so we don't depend on system tools — the ephemeral
# self-hosted runner doesn't have gh/jq installed.
- name: Resolve PR head
if: github.event_name == 'issue_comment'
id: pr
uses: actions/github-script@v7
with:
script: |
const pr = await github.rest.pulls.get({
owner: context.repo.owner,
repo: context.repo.repo,
pull_number: context.issue.number,
});
// Even though the commenter has write access, the PR's code
// lives on the head repo. If that's a fork, we'd be running
// untrusted code on the self-hosted runner.
const headRepo = pr.data.head.repo.full_name;
const expected = `${context.repo.owner}/${context.repo.repo}`;
if (headRepo !== expected) {
core.setFailed(`PR #${context.issue.number} is from a fork (${headRepo}); /pytest is not supported for forks.`);
return;
}
core.setOutput('head_sha', pr.data.head.sha);
core.setOutput('base_ref', pr.data.base.ref);
- name: Resolve tested revision identity
id: identity
env:
EVENT_NAME: ${{ github.event_name }}
COMMENT_HEAD_SHA: ${{ steps.pr.outputs.head_sha }}
EVENT_SHA: ${{ github.sha }}
COMMENT_PR_NUMBER: ${{ github.event.issue.number }}
EVENT_PR_NUMBER: ${{ github.event.pull_request.number }}
run: |
if [[ "$EVENT_NAME" == "issue_comment" ]]; then
if [[ -z "$COMMENT_HEAD_SHA" ]]; then
echo "::error::Refusing /pytest run: PR head SHA was not resolved."
exit 1
fi
echo "tested_sha=$COMMENT_HEAD_SHA" >> "$GITHUB_OUTPUT"
echo "pr_number=$COMMENT_PR_NUMBER" >> "$GITHUB_OUTPUT"
else
echo "tested_sha=$EVENT_SHA" >> "$GITHUB_OUTPUT"
echo "pr_number=$EVENT_PR_NUMBER" >> "$GITHUB_OUTPUT"
fi
# Parsed up-front (before checkout) so the acknowledgment comment below
# can echo the resolved args. This step only reads env vars, so it
# doesn't need the working tree.
- name: Parse pytest args
id: parse
env:
GH_EVENT_NAME: ${{ github.event_name }}
COMMENT_BODY: ${{ github.event.comment.body }}
INPUT_MARKS: ${{ inputs.marks }}
INPUT_SIM: ${{ inputs.sim }}
INPUT_NUM_ROBOTS: ${{ inputs.num_robots }}
INPUT_ITERATIONS: ${{ inputs.stress_iterations }}
INPUT_STABLE: ${{ inputs.stable_duration }}
INPUT_TRAJECTORIES: ${{ inputs.trajectory_types }}
INPUT_TAKEOFF_VELOCITIES: ${{ inputs.takeoff_velocities }}
run: |
python3 <<'PYEOF'
import os, shlex, sys
event = os.environ['GH_EVENT_NAME']
if event == 'workflow_dispatch':
args = []
if (m := os.environ.get('INPUT_MARKS', '').strip()):
args.extend(['-m', m])
if (s := os.environ.get('INPUT_SIM', '').strip()):
args.extend(['--sim', s])
if (n := os.environ.get('INPUT_NUM_ROBOTS', '').strip()):
args.extend(['--num-robots', n])
if (it := os.environ.get('INPUT_ITERATIONS', '').strip()):
args.extend(['--stress-iterations', it])
if (st := os.environ.get('INPUT_STABLE', '').strip()):
args.extend(['--stable-duration', st])
if (trajectories := os.environ.get('INPUT_TRAJECTORIES', '').strip()):
args.extend(['--trajectory-types', trajectories])
if (velocities := os.environ.get('INPUT_TAKEOFF_VELOCITIES', '').strip()):
args.extend(['--takeoff-velocities', velocities])
elif event == 'pull_request':
# Automatic PR validation is deliberately build-scoped. Fast
# Python unit tests run in unit-tests.yml; GPU simulation remains
# selectable via /pytest or workflow_dispatch.
args = ['-m', 'build_packages']
else:
body = os.environ.get('COMMENT_BODY', '')
# Only the first line is parsed — everything below it is
# treated as freeform comment text (notes, context, etc.).
first_line = (body.splitlines()[0] if body else '').strip()
if not first_line.startswith('/pytest'):
print('::error::Comment does not start with /pytest', file=sys.stderr)
sys.exit(1)
args_line = first_line[len('/pytest'):].strip()
try:
args = shlex.split(args_line)
except ValueError as e:
print(f'::error::Could not parse pytest args from comment: {e}', file=sys.stderr)
sys.exit(1)
# CI-only flag: do not forward to pytest.
no_image_build = False
stripped = []
for a in args:
if a in ('--no-image-build', '--pull-only'):
no_image_build = True
else:
stripped.append(a)
args = stripped
# Pull out --sim and -m so the image-prep step can scope profiles
# and decide whether to skip (build_docker tests rebuild themselves).
# When --sim isn't given we mirror conftest's default so prep covers
# whatever pytest will actually exercise.
sim = 'isaacsim'
sim_explicit = False
marks = ''
marks_idx = -1
for i, a in enumerate(args):
if a == '--sim' and i + 1 < len(args):
sim = args[i + 1]
sim_explicit = True
elif a == '-m' and i + 1 < len(args):
marks = args[i + 1]
marks_idx = i + 1
# When the user specified any marks, prepend build_packages so code
# is built before launch tests try to use it. Skipped when no marks
# are given (pytest runs everything including build_packages) and
# when build_packages is already in the expression.
if marks and 'build_packages' not in marks:
marks = f'build_packages or {marks}'
args[marks_idx] = marks
# colcon tests do not need a sim image. Default --sim would otherwise
# bake isaac-sim before a 1s colcon test. Pull registry cache tags
# instead; never image-build.
marks_norm = marks.replace('"', '').replace("'", '').strip()
args_blob = ' '.join(args)
heavy = any(m in marks_norm for m in (
'liveliness', 'sensors', 'takeoff_hover_land', 'autonomy', 'build_docker',
'optitrack',
))
only_packages = marks_norm == 'build_packages' or (
not heavy and any(s in args_blob for s in (
'test_build_packages', 'test_colcon_',
))
)
if only_packages:
no_image_build = True
if not sim_explicit:
sim = 'msairsim'
skip_prep = 'build_docker' in marks
quoted = ' '.join(shlex.quote(a) for a in args)
with open(os.environ['GITHUB_OUTPUT'], 'a') as f:
f.write(f'pytest_args={quoted}\n')
f.write(f'sim={sim}\n')
f.write(f'skip_image_prep={"true" if skip_prep else "false"}\n')
f.write(f'no_image_build={"true" if no_image_build else "false"}\n')
print(f'Resolved pytest args: {quoted or "(none — pytest defaults)"}')
print(f'Resolved sim profile: {sim}')
print(f'Skip image prep: {skip_prep}')
print(f'No image build (pull/retag only): {no_image_build}')
PYEOF
# Reply on the PR thread so the commenter sees their /pytest was
# picked up and can confirm we parsed the args correctly. The
# workflow_dispatch path skips this (no PR to comment on); the
# pull_request path skips it too (the PR Checks tab is
# already showing the native run).
- name: Post acknowledgment comment
if: github.event_name == 'issue_comment'
uses: actions/github-script@v7
with:
script: |
const args = ${{ toJSON(steps.parse.outputs.pytest_args) }};
const cmd = `pytest tests/ ${args}`.trim();
const runUrl = `${context.serverUrl}/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}`;
const pullOnly = '${{ steps.parse.outputs.no_image_build }}' === 'true';
const note = pullOnly
? `Note: pull-only image prep (no \`image-build\`). \`-m build_packages\` does not pull Isaac Sim. Add \`--no-image-build\` on other marks to skip rebuilds.`
: `Note: \`build_packages\` is automatically prepended whenever any marks are specified, to ensure code is built before launch tests run.`;
await github.rest.issues.createComment({
owner: context.repo.owner,
repo: context.repo.repo,
issue_number: context.issue.number,
body: `Running \`${cmd}\` — [view run](${runUrl}). Status will appear as a check on this PR.\n\n${note}`,
});
# Pin a Check Run on the PR's head SHA so the run shows up in the
# PR's "Checks" tab while it executes — issue_comment-triggered runs
# are otherwise associated with the default branch and don't surface
# on the PR. Finalized at end of job with the actual conclusion.
- name: Open in-progress check on PR head
if: github.event_name == 'issue_comment'
id: check_create
uses: actions/github-script@v7
with:
script: |
const res = await github.rest.checks.create({
owner: context.repo.owner,
repo: context.repo.repo,
name: 'System Tests',
head_sha: '${{ steps.pr.outputs.head_sha }}',
status: 'in_progress',
details_url: `${context.serverUrl}/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}`,
});
core.setOutput('id', res.data.id);
- name: Checkout
uses: actions/checkout@v4
with:
# For issue_comment we must check out the PR head explicitly —
# GITHUB_SHA points at the default branch for that event. For
# workflow_dispatch and pull_request, empty string lets checkout
# use its default (the dispatched ref / the PR merge commit).
ref: ${{ github.event_name == 'issue_comment' && steps.pr.outputs.head_sha || '' }}
submodules: recursive
- name: Create Isaac Sim omni_pass.env
run: |
mkdir -p simulation/isaac-sim/docker
cat > simulation/isaac-sim/docker/omni_pass.env <<'EOF'
OMNI_USER=guest
OMNI_PASS=guest
OMNI_SERVER="omniverse://airlab-nucleus.andrew.cmu.edu/NVIDIA/Assets/Isaac/5.1"
ACCEPT_EULA=Y
OMNI_ENV_PRIVACY_CONSENT=Y
EOF
- name: Install test dependencies
# Ubuntu 24.04 marks the system Python as externally-managed (PEP 668),
# so `pip install` outside a venv is rejected. Use a venv and prepend
# its bin/ to $GITHUB_PATH so subsequent steps pick up `pytest`
# automatically.
run: |
sudo apt-get update -qq
sudo apt-get install -y --no-install-recommends python3-venv
python3 -m venv .venv
echo "$GITHUB_WORKSPACE/.venv/bin" >> "$GITHUB_PATH"
.venv/bin/pip install --upgrade pip
.venv/bin/pip install -r tests/requirements.txt
# Optional registry-cache mode. When the secrets/vars are present we log
# in to the internal Docker registry; the next step then sets
# AIRSTACK_REGISTRY_CACHE=1 so airstack.sh pre-pulls + uses BuildKit
# inline cache (build_docker tests get layer-reuse speedup) and pre-pulls
# before `airstack up` (other tests skip the implicit rebuild). When
# secrets are absent both steps are skipped and behavior is unchanged.
#
# Read-only on purpose: AIRSTACK_REGISTRY_CACHE_PUSH stays unset here so a
# PR can consume the floating cache tag but never republish it. Only
# docker-build.yml (main/develop) writes it.
- name: Log in to internal Docker registry
id: docker_login
if: ${{ vars.DOCKER_REGISTRY_URL != '' && env.DOCKER_REGISTRY_PASSWORD != '' }}
uses: docker/login-action@v3
with:
registry: ${{ vars.DOCKER_REGISTRY_URL }}
username: ${{ vars.DOCKER_REGISTRY_USERNAME }}
password: ${{ secrets.DOCKER_REGISTRY_PASSWORD }}
- name: Enable registry-cache mode
if: ${{ steps.docker_login.outcome == 'success' }}
run: echo "AIRSTACK_REGISTRY_CACHE=1" >> "$GITHUB_ENV"
- name: Ensure airstack.sh is executable
run: chmod +x airstack.sh
- name: Disable compose image builds
if: ${{ steps.parse.outputs.no_image_build == 'true' }}
run: echo "AIRSTACK_NO_IMAGE_BUILD=1" >> "$GITHUB_ENV"
# The ephemeral runner starts with no local images. `airstack_env` in
# tests/conftest.py fails fast if compose images are missing, so prep
# them here. Profile-gated services (ms-airsim, isaac-sim) are skipped
# by compose unless their profile is active, so we mirror the fixture's
# profile selection from the parsed --sim. Pull versioned tags, then
# retag floating cache_* tags onto the VERSION name (PR tags never
# exist). Fall back to image-build only when --no-image-build is off.
# Skipped when marks contain build_docker — those tests build themselves.
- name: Ensure Docker images present
if: ${{ steps.parse.outputs.skip_image_prep != 'true' }}
env:
AIRSTACK_ROOT: ${{ github.workspace }}
SIM_INPUT: ${{ steps.parse.outputs.sim }}
NO_IMAGE_BUILD: ${{ steps.parse.outputs.no_image_build }}
run: |
profiles=desktop
[[ ",$SIM_INPUT," == *,msairsim,* ]] && profiles="$profiles,ms-airsim"
[[ ",$SIM_INPUT," == *,isaacsim,* ]] && profiles="$profiles,isaac-sim"
export COMPOSE_PROFILES="$profiles"
echo "Pulling images for COMPOSE_PROFILES=$COMPOSE_PROFILES (no_image_build=$NO_IMAGE_BUILD)"
# Pull from registry; tolerate per-image failures so we can detect
# what's still missing afterwards instead of aborting on the first
# gap. `--progress=quiet` suppresses per-layer progress; errors
# still surface on stderr.
./airstack.sh --progress=quiet images pull --ignore-pull-failures || true
# VERSION tags miss on every PR. Seed from floating cache_* tags.
cache_tag="$(grep -E '^CACHE_TAG=' .env 2>/dev/null | cut -d= -f2 | tr -d '"' || true)"
cache_tag="${cache_tag:-cache}"
while IFS= read -r img; do
[[ -z "$img" ]] && continue
if docker image inspect "$img" --format '{{.Id}}' >/dev/null 2>&1; then
continue
fi
# Replace :v<semver>_ with :<cache_tag>_ (PR versioned tags never exist)
cache_img="$(python3 -c "import re,sys; print(re.sub(r':v[^_]+_', f':{sys.argv[2]}_', sys.argv[1], count=1))" "$img" "$cache_tag")"
echo "Versioned tag missing; trying cache tag $cache_img"
if docker pull --quiet "$cache_img"; then
docker tag "$cache_img" "$img"
echo "Retagged $cache_img -> $img"
else
echo "Cache tag pull failed for $cache_img"
fi
done < <(docker compose -f docker-compose.yaml config --images)
missing=()
while IFS= read -r img; do
[[ -z "$img" ]] && continue
if ! docker image inspect "$img" --format '{{.Id}}' >/dev/null 2>&1; then
missing+=("$img")
fi
done < <(docker compose -f docker-compose.yaml config --images)
if (( ${#missing[@]} > 0 )); then
echo "Images still missing after pull/retag:"
printf ' - %s\n' "${missing[@]}"
if [[ "$NO_IMAGE_BUILD" == "true" ]]; then
echo "::error::Pull-only mode (--no-image-build or -m build_packages) will not run 'images build'. Run /pytest -m build_docker once, or omit --no-image-build."
exit 1
fi
echo "Falling back to images build"
./airstack.sh --progress=quiet images build
else
echo "All required images present after pull/retag — skipping build."
fi
- name: Run tests
env:
AIRSTACK_ROOT: ${{ github.workspace }}
AIRSTACK_TESTED_SHA: ${{ steps.identity.outputs.tested_sha }}
AIRSTACK_PR_NUMBER: ${{ steps.identity.outputs.pr_number }}
DISPLAY: ""
PYTEST_ARGS: ${{ steps.parse.outputs.pytest_args }}
run: |
# Re-split the shell-quoted args from the parse step so we forward
# them to pytest as a proper argv list (preserving values like
# `-m 'a or b'`). sys.stdout.write is intentional: print('') emits
# one blank line, which mapfile turns into an empty positional path
# and makes pytest recurse from the repository root.
mapfile -t ARGS < <(python3 -c "import os, shlex, sys; sys.stdout.write(''.join(f'{arg}\\n' for arg in shlex.split(os.environ['PYTEST_ARGS'])))")
for arg in "${ARGS[@]}"; do
if [[ -z "$arg" ]]; then
echo "::error::Refusing an empty pytest argument because it expands collection to the repository root."
exit 2
fi
done
set +e
pytest tests/ \
"${ARGS[@]}" \
-v -s \
--log-cli-level=INFO \
--log-cli-format='%(asctime)s [%(levelname)s] %(name)s: %(message)s' \
--log-cli-date-format='%H:%M:%S'
pytest_status=$?
set -e
if (( pytest_status != 0 )); then
exit "$pytest_status"
fi
# A successful collect-only/non-executed campaign is not a passing
# system test. Make that distinction visible in the job conclusion.
python3 <<'PYEOF'
import json
from pathlib import Path
candidates = list(Path("tests/results").glob("*/run_meta.json"))
if not candidates:
raise SystemExit("::error::pytest succeeded without run_meta.json")
latest = max(candidates, key=lambda path: path.stat().st_mtime)
outcome = json.loads(latest.read_text()).get("outcome")
if outcome not in {"simulation", "non_simulation"}:
raise SystemExit(
f"::error::pytest did not execute a complete campaign ({outcome})"
)
PYEOF
- name: Upload test results
uses: actions/upload-artifact@v4
if: always()
with:
name: test-results-${{ steps.identity.outputs.tested_sha }}-${{ github.run_id }}
path: tests/results/
retention-days: 90
report:
name: Metrics Report
runs-on: ubuntu-latest
needs: run-tests
# Skip when run-tests was skipped (e.g., comment didn't match `/pytest`)
# so we don't post empty-report comments on every PR comment.
if: >
always() &&
needs.run-tests.result != 'skipped' &&
needs.run-tests.outputs.tested_sha != '' &&
(needs.run-tests.result != 'cancelled' || github.event_name != 'pull_request')
permissions:
actions: read
checks: write
contents: read
pull-requests: write
steps:
- name: Checkout
uses: actions/checkout@v4
with:
ref: ${{ needs.run-tests.outputs.tested_sha }}
- name: Set up Python
uses: actions/setup-python@v5
with:
python-version: "3.12"
- name: Install report dependencies
# parse_metrics.py imports the tests/harness package, so it needs the
# full test requirements (pyyaml et al.), not just tabulate.
run: pip install -r tests/requirements.txt
- name: Resolve PR base branch
if: github.event_name == 'issue_comment' || github.event_name == 'pull_request'
id: pr_ctx
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: |
if [ "${{ github.event_name }}" = "pull_request" ]; then
echo "base_ref=${{ github.base_ref }}" >> "$GITHUB_OUTPUT"
else
BASE=$(gh api repos/${{ github.repository }}/pulls/${{ github.event.issue.number }} --jq .base.ref)
echo "base_ref=$BASE" >> "$GITHUB_OUTPUT"
fi
- name: Download current test results
uses: actions/download-artifact@v4
continue-on-error: true
with:
name: test-results-${{ needs.run-tests.outputs.tested_sha }}-${{ github.run_id }}
path: current-results/
# PR mode (opened or comment-triggered): fetch latest artifact from
# the PR's base branch (e.g. develop or main).
- name: Download baseline results (PR)
if: github.event_name == 'issue_comment' || github.event_name == 'pull_request'
env:
GH_TOKEN: ${{ github.token }}
BASE_REF: ${{ steps.pr_ctx.outputs.base_ref }}
run: |
mkdir -p baseline-results
gh api --method GET \
"repos/${{ github.repository }}/actions/workflows/system-tests.yml/runs" \
-f branch="$BASE_REF" -f status=success -f per_page=20 \
--jq '.workflow_runs[].id' |
while read -r run_id; do
gh run download "$run_id" --repo "${{ github.repository }}" \
--pattern "test-results-*" --dir "baseline-results/$run_id" || true
done
# Manual dispatch with explicit baseline run ID
- name: Download baseline results (manual, explicit run ID)
if: >
github.event_name == 'workflow_dispatch' &&
inputs.baseline_run_id != ''
uses: actions/download-artifact@v4
continue-on-error: true
with:
github-token: ${{ github.token }}
run-id: ${{ inputs.baseline_run_id }}
pattern: "test-results-*"
merge-multiple: true
path: baseline-results/
# Manual dispatch without explicit baseline: fetch latest from main
- name: Download baseline results (manual, latest main)
if: >
github.event_name == 'workflow_dispatch' &&
inputs.baseline_run_id == ''
env:
GH_TOKEN: ${{ github.token }}
run: |
mkdir -p baseline-results
gh api --method GET \
"repos/${{ github.repository }}/actions/workflows/system-tests.yml/runs" \
-f branch=main -f status=success -f per_page=20 \
--jq '.workflow_runs[].id' |
while read -r run_id; do
gh run download "$run_id" --repo "${{ github.repository }}" \
--pattern "test-results-*" --dir "baseline-results/$run_id" || true
done
- name: Locate result directories
id: dirs
# Find the current dir, then choose a baseline from the recent-run
# candidate tree by completed campaign fingerprint.
run: |
CURRENT_XML=$(find current-results/ -name results.xml 2>/dev/null | sort -r | head -1)
[ -n "$CURRENT_XML" ] && echo "current=$(dirname "$CURRENT_XML")" >> "$GITHUB_OUTPUT"
if [ -n "$CURRENT_XML" ] && [ -d baseline-results ]; then
CURRENT_DIR="$(dirname "$CURRENT_XML")"
BASELINE=$(PYTHONPATH=tests python3 - "$CURRENT_DIR" <<'PYEOF'
import sys
from pathlib import Path
from harness.baseline import select_baseline_path
selected = select_baseline_path(Path(sys.argv[1]), Path("baseline-results"))
print(selected or "")
PYEOF
)
echo "baseline=$BASELINE" >> "$GITHUB_OUTPUT"
else
echo "baseline=" >> "$GITHUB_OUTPUT"
fi
- name: Generate metrics report
id: report
continue-on-error: true
run: |
CURRENT="${{ steps.dirs.outputs.current }}"
BASELINE="${{ steps.dirs.outputs.baseline }}"
if [ -z "$CURRENT" ]; then
cat > report.md <<'EOF'
## Run status
**Simulation metrics are not comparable.** The test job produced no finalized `results.xml` artifact. The runner may have timed out, been cancelled, or failed before pytest started.
Pass-rate and regression tables are suppressed because no completed test campaign is available.
EOF
echo "parser_exit=2" >> "$GITHUB_OUTPUT"
exit 2
fi
set +e
if [ -n "$BASELINE" ]; then
python tests/parse_metrics.py \
--current "$CURRENT" \
--baseline "$BASELINE" \
--output report.md
else
python tests/parse_metrics.py \
--current "$CURRENT" \
--output report.md
fi
parser_exit=$?
set -e
echo "parser_exit=$parser_exit" >> "$GITHUB_OUTPUT"
exit "$parser_exit"
- name: Post PR comment
if: github.event_name == 'issue_comment' || github.event_name == 'pull_request'
uses: actions/github-script@v7
with:
script: |
const fs = require('fs');
let body;
try {
body = fs.readFileSync('report.md', 'utf8');
} catch {
body = '_No metrics report generated._';
}
const header = `## Test Metrics — \`${{ needs.run-tests.outputs.tested_sha }}\`\n\n`;
await github.rest.issues.createComment({
issue_number: Number('${{ needs.run-tests.outputs.pr_number }}'),
owner: context.repo.owner,
repo: context.repo.repo,
body: header + body,
});
- name: Write job summary
if: always()
run: |
if [ -f report.md ]; then
echo "## Test Metrics — \`${{ needs.run-tests.outputs.tested_sha }}\`" >> "$GITHUB_STEP_SUMMARY"
echo "" >> "$GITHUB_STEP_SUMMARY"
cat report.md >> "$GITHUB_STEP_SUMMARY"
else
echo "_No metrics report generated._" >> "$GITHUB_STEP_SUMMARY"
fi
- name: Fail on report integrity error
if: steps.report.outcome == 'failure'
# Numeric metric deltas are advisory (parse_metrics.py exits 0 on
# them); this step fires only on parser/integrity failures (exit 2)
# or an uncaught crash, and never labels either a metric regression.
run: |
echo "::error::Metrics report generation failed — see the report step log."
exit 1
- name: Finalize check on PR head
if: always() && github.event_name == 'issue_comment' && needs.run-tests.outputs.check_run_id
uses: actions/github-script@v7
with:
script: |
const tests = '${{ needs.run-tests.result }}';
const report = '${{ steps.report.outcome }}';
let conclusion = 'failure';
if (tests === 'success' && report === 'success') {
conclusion = 'success';
} else if (tests === 'cancelled') {
conclusion = 'cancelled';
}
await github.rest.checks.update({
owner: context.repo.owner,
repo: context.repo.repo,
check_run_id: Number('${{ needs.run-tests.outputs.check_run_id }}'),
status: 'completed',
conclusion,
details_url: `${context.serverUrl}/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}`,
});