Files
miles/.github/workflows/pr-test.yml
T

365 lines
16 KiB
YAML

# doc-dev: docs/developer/ci/00-stage.md
# doc-dev: docs/developer/ci/01-label.md
name: PR Test
on:
pull_request:
types: [opened, synchronize, reopened, ready_for_review, labeled, closed]
schedule:
# All scheduled runs use UTC. Saturday's weekly full CI replaces nightly.
- cron: '0 15 * * 0-5'
- cron: '0 15 * * 6'
workflow_dispatch:
inputs:
infinite_run:
description: 'Run training infinitely'
required: false
type: boolean
default: false
# Empty defaults are load-bearing: a non-empty default would count as an
# explicit override and shadow release-lock.json on a release branch
# (precedence: explicit input / PR body > lockfile > branch HEAD).
ci_megatron_pr:
description: 'Megatron-LM branch/commit (empty: release-lock.json if present, else miles-main)'
required: false
type: string
default: ''
ci_sglang_pr:
description: 'SGLang branch/commit (empty: release-lock.json if present, else sglang-miles)'
required: false
type: string
default: ''
ci_image_tag:
description: 'Miles Docker image tag (default: dev)'
required: false
type: string
default: 'dev'
workflow_call:
inputs:
ref:
description: 'Git ref (branch, tag, or SHA) to test, e.g. release/v0.3.0.'
required: true
type: string
cadence:
description: 'Explicit CI cadence (regular/nightly/weekly/release). Release cuts use release: weekly scope, no baseline writes.'
required: false
type: string
default: 'release'
image_tag:
description: 'Miles Docker image tag to run suites in (e.g. a frozen dev-<timestamp> tag).'
required: false
type: string
default: ''
permissions:
contents: read
concurrency:
# PR updates supersede their previous run; distinct cron schedules and
# manual operations must not cancel one another on the default branch.
# The prefix is a literal, not github.workflow: in a called workflow that
# context resolves to the *caller's* name, which would collapse the CUDA
# and ROCm groups of one release-branch-cut run into the same group and
# let them cancel each other. inputs.ref (workflow_call only) groups
# release CI by branch, so a ② re-dispatch cancels the stale previous run.
group: pr-test-${{ github.event.number || github.event.schedule || inputs.ref || github.run_id }}
cancel-in-progress: true
# `resolve-ci-policy` passes trigger-specific facts to `ci_policy.py`, which
# owns the explicit cadence, raw labels, test scope, stage selection, and
# fast-fail policy.
# `run_suite.py` consumes that shared policy; GPU jobs consume the shared
# skip list and bypass output for their job-level gates.
#
# Trigger type is not policy: each scheduled cron maps to an explicit cadence,
# while workflow_dispatch remains a regular manual operation with no implicit
# domain scope.
jobs:
resolve-ci-policy:
# A PR based on another branch (e.g. a stacked PR) starts CI only while it
# carries a `run-ci*` label; `"run-ci` in the JSON array matches a label
# prefix. Every other job needs this one, so skipping it skips the whole run.
if: |
github.event.action != 'closed' &&
(github.event_name != 'pull_request' ||
github.event.pull_request.base.ref == github.event.repository.default_branch ||
contains(toJSON(github.event.pull_request.labels.*.name), '"run-ci'))
runs-on: ubuntu-latest
outputs:
cadence: ${{ steps.resolve.outputs.cadence }}
raw_labels: ${{ steps.resolve.outputs.raw_labels }}
bypass_fastfail: ${{ steps.resolve.outputs.bypass_fastfail }}
skipped_stages: ${{ steps.resolve.outputs.skipped_stages }}
needs_cuda_image: ${{ steps.resolve.outputs.needs_cuda_image }}
steps:
- name: Checkout repository
# Deliberately not inputs.ref: this job executes ci_policy.py, so it
# always runs the triggering commit's copy (CodeQL cache-poisoning
# surface). The cadence arrives via CADENCE_OVERRIDE; only the suite
# jobs check out the release ref.
uses: actions/checkout@v4
with:
fetch-depth: 2
persist-credentials: false
- name: Collect PR changed files
if: github.event_name == 'pull_request'
shell: bash
run: |
if ! git diff --name-status -z -M HEAD^1 HEAD > "$RUNNER_TEMP/changed-files.z"; then
rm -f "$RUNNER_TEMP/changed-files.z"
echo "::warning::Unable to compute PR diff; all selected GPU stages are treated as affected."
fi
- name: Resolve CI policy inputs
id: resolve
env:
EVENT_NAME: ${{ github.event_name }}
SCHEDULE: ${{ github.event.schedule || '' }}
PR_LABELS_JSON: ${{ toJSON(github.event.pull_request.labels.*.name) }}
# workflow_call inherits the caller's event_name, so the cadence must
# be passed explicitly rather than inferred from the trigger.
CADENCE_OVERRIDE: ${{ inputs.cadence || '' }}
CHANGED_FILES_PATH: ${{ runner.temp }}/changed-files.z
run: python -m tests.ci.ci_policy
# Selected CUDA tests need an image after the fast CPU gate completes.
# Nightly, weekly, and bypass-fastfail preserve their cross-stage exception.
docker-build:
needs: [resolve-ci-policy, stage-a-cpu]
if: |
always() && !cancelled() &&
github.event.action != 'closed' &&
needs.resolve-ci-policy.result == 'success' &&
needs.resolve-ci-policy.outputs.needs_cuda_image == 'true' &&
(needs.stage-a-cpu.result == 'success' ||
needs.resolve-ci-policy.outputs.bypass_fastfail == 'true')
# Scoped to this call: a build consumes the one-shot rebuild-ci-image label by
# removing it. Every other job keeps the workflow-level read-only token.
permissions:
contents: read
actions: read
pull-requests: write
uses: ./.github/workflows/_build-pr-ci-image.yml
secrets: inherit
# Gate on this build alone. Without an `if` the default success() spans the whole
# dependency closure, so a stage-a-cpu failure skipped this and every GPU stage even
# on a nightly, whose policy sets bypass_fastfail precisely to keep them running.
resolve-ci-image:
needs: [docker-build]
if: always() && !cancelled() && needs.docker-build.result == 'success'
runs-on: ubuntu-latest
outputs:
container_image: ${{ steps.resolve.outputs.container_image }}
steps:
- name: Resolve CI image
id: resolve
shell: bash
env:
PR_BODY: ${{ github.event.pull_request.body || '' }}
# workflow_call image_tag (a frozen release-CI tag) outranks the
# dispatch input, which only exists on manual runs anyway.
INPUT_CI_IMAGE_TAG: ${{ inputs.image_tag || github.event.inputs.ci_image_tag || '' }}
# Not `built`: a run that reused an unchanged image published earlier must still
# test against that image rather than silently falling back to the released one.
PR_IMAGE_AVAILABLE: ${{ needs.docker-build.outputs.tag_available }}
PR_NUMBER: ${{ github.event.pull_request.number || '' }}
run: |
CI_IMAGE_TAG="${INPUT_CI_IMAGE_TAG}"
[ -z "$CI_IMAGE_TAG" ] && [ "$PR_IMAGE_AVAILABLE" = "true" ] && CI_IMAGE_TAG="pr-${PR_NUMBER}"
if [ -n "$PR_BODY" ]; then
PR_CI_IMAGE_TAG=$(echo "$PR_BODY" | grep -m1 -oP '^ci-image-tag:\s+\K\S+' || true)
[ -z "$CI_IMAGE_TAG" ] && [ -n "$PR_CI_IMAGE_TAG" ] && CI_IMAGE_TAG="$PR_CI_IMAGE_TAG"
fi
[ -z "$CI_IMAGE_TAG" ] && CI_IMAGE_TAG="dev"
if [[ ! "$CI_IMAGE_TAG" =~ ^[A-Za-z0-9_][A-Za-z0-9_.-]{0,127}$ ]]; then
echo "::error::ci-image-tag must be a Docker tag, not a full image name: $CI_IMAGE_TAG"
exit 1
fi
echo "container_image=radixark/miles:${CI_IMAGE_TAG}" >> "$GITHUB_OUTPUT"
echo "Resolved CI image: radixark/miles:${CI_IMAGE_TAG}"
# Stage A: CPU-only fast tests (runs whenever resolve-ci-policy runs)
# Uses the CPU-only reusable workflow on GitHub-hosted ubuntu-latest, so
# fast tests do not occupy GPU-fleet runner slots.
stage-a-cpu:
needs: [resolve-ci-policy]
strategy:
fail-fast: false
matrix:
partition_id: [0, 1, 2, 3]
uses: ./.github/workflows/_run-cpu-ci.yml
with:
ref: ${{ inputs.ref || '' }}
execute_command: >-
python tests/ci/run_suite.py --hw cpu --suite stage-a-cpu
--auto-partition-id ${{ matrix.partition_id }} --auto-partition-size 4
--cadence ${{ needs.resolve-ci-policy.outputs.cadence }}
--labels ${{ needs.resolve-ci-policy.outputs.raw_labels }}
secrets: inherit
# Stage B: CPU-only bucket for slower CPU tests that don't fit stage-a-cpu's
# fast budget. Runs whenever resolve-ci-policy runs; an empty suite still exits 0
# in run_suite.py.
stage-b-cpu:
needs: [resolve-ci-policy]
uses: ./.github/workflows/_run-cpu-ci.yml
with:
ref: ${{ inputs.ref || '' }}
execute_command: >-
python tests/ci/run_suite.py --hw cpu --suite stage-b-cpu
--cadence ${{ needs.resolve-ci-policy.outputs.cadence }}
--labels ${{ needs.resolve-ci-policy.outputs.raw_labels }}
secrets: inherit
stage-b-2-gpu-h200:
needs: [resolve-ci-policy, resolve-ci-image, stage-a-cpu]
if: |
always() && !cancelled() &&
needs.resolve-ci-policy.result == 'success' &&
needs.resolve-ci-image.result == 'success' &&
!contains(fromJSON(needs.resolve-ci-policy.outputs.skipped_stages || '[]'), 'stage-b-2-gpu-h200') &&
(needs.stage-a-cpu.result == 'success' ||
needs.resolve-ci-policy.outputs.bypass_fastfail == 'true')
uses: ./.github/workflows/_run-ci.yml
with:
runs_on: '["h200", "2gpu"]'
container_image: ${{ needs.resolve-ci-image.outputs.container_image }}
ref: ${{ inputs.ref || '' }}
execute_command: >-
python tests/ci/run_suite.py --hw cuda --suite stage-b-2-gpu-h200
--cadence ${{ needs.resolve-ci-policy.outputs.cadence }}
--labels ${{ needs.resolve-ci-policy.outputs.raw_labels }}
${{ github.event_name == 'workflow_dispatch' && '--match-all-labels' || '' }}
secrets: inherit
# No partition matrix: the tests left on 8x H100 fit one runner's run, so one
# job keeps the second H100 host free for other runs.
stage-c-8-gpu-h100:
needs: [resolve-ci-policy, resolve-ci-image, stage-a-cpu]
if: |
always() && !cancelled() &&
needs.resolve-ci-policy.result == 'success' &&
needs.resolve-ci-image.result == 'success' &&
!contains(fromJSON(needs.resolve-ci-policy.outputs.skipped_stages || '[]'), 'stage-c-8-gpu-h100') &&
(needs.stage-a-cpu.result == 'success' ||
needs.resolve-ci-policy.outputs.bypass_fastfail == 'true')
uses: ./.github/workflows/_run-ci.yml
with:
runs_on: '["h100", "8gpu"]'
container_image: ${{ needs.resolve-ci-image.outputs.container_image }}
ref: ${{ inputs.ref || '' }}
execute_command: >-
python tests/ci/run_suite.py --hw cuda --suite stage-c-8-gpu-h100
--cadence ${{ needs.resolve-ci-policy.outputs.cadence }}
--labels ${{ needs.resolve-ci-policy.outputs.raw_labels }}
${{ github.event_name == 'workflow_dispatch' && '--match-all-labels' || '' }}
secrets: inherit
stage-c-8-gpu-h200:
needs: [resolve-ci-policy, resolve-ci-image, stage-a-cpu]
if: |
always() && !cancelled() &&
needs.resolve-ci-policy.result == 'success' &&
needs.resolve-ci-image.result == 'success' &&
!contains(fromJSON(needs.resolve-ci-policy.outputs.skipped_stages || '[]'), 'stage-c-8-gpu-h200') &&
(needs.stage-a-cpu.result == 'success' ||
needs.resolve-ci-policy.outputs.bypass_fastfail == 'true')
strategy:
fail-fast: false
max-parallel: ${{ needs.resolve-ci-policy.outputs.cadence == 'weekly' && 1 || 2 }}
matrix:
partition_id: [0, 1]
uses: ./.github/workflows/_run-ci.yml
with:
runs_on: '["h200", "8gpu"]'
container_image: ${{ needs.resolve-ci-image.outputs.container_image }}
ref: ${{ inputs.ref || '' }}
execute_command: >-
python tests/ci/run_suite.py --hw cuda --suite stage-c-8-gpu-h200
--auto-partition-id ${{ matrix.partition_id }} --auto-partition-size 2
--cadence ${{ needs.resolve-ci-policy.outputs.cadence }}
--labels ${{ needs.resolve-ci-policy.outputs.raw_labels }}
${{ github.event_name == 'workflow_dispatch' && '--match-all-labels' || '' }}
secrets: inherit
stage-c-4-gpu-h200:
needs: [resolve-ci-policy, resolve-ci-image, stage-a-cpu]
if: |
always() && !cancelled() &&
needs.resolve-ci-policy.result == 'success' &&
needs.resolve-ci-image.result == 'success' &&
!contains(fromJSON(needs.resolve-ci-policy.outputs.skipped_stages || '[]'), 'stage-c-4-gpu-h200') &&
(needs.stage-a-cpu.result == 'success' ||
needs.resolve-ci-policy.outputs.bypass_fastfail == 'true')
strategy:
fail-fast: false
max-parallel: ${{ needs.resolve-ci-policy.outputs.cadence == 'weekly' && 1 || 3 }}
matrix:
partition_id: [0, 1, 2]
uses: ./.github/workflows/_run-ci.yml
with:
runs_on: '["h200", "4gpu"]'
container_image: ${{ needs.resolve-ci-image.outputs.container_image }}
ref: ${{ inputs.ref || '' }}
execute_command: >-
python tests/ci/run_suite.py --hw cuda --suite stage-c-4-gpu-h200
--auto-partition-id ${{ matrix.partition_id }} --auto-partition-size 3
--cadence ${{ needs.resolve-ci-policy.outputs.cadence }}
--labels ${{ needs.resolve-ci-policy.outputs.raw_labels }}
${{ github.event_name == 'workflow_dispatch' && '--match-all-labels' || '' }}
secrets: inherit
stage-c-2-gpu-h200:
needs: [resolve-ci-policy, resolve-ci-image, stage-a-cpu]
if: |
always() && !cancelled() &&
needs.resolve-ci-policy.result == 'success' &&
needs.resolve-ci-image.result == 'success' &&
!contains(fromJSON(needs.resolve-ci-policy.outputs.skipped_stages || '[]'), 'stage-c-2-gpu-h200') &&
(needs.stage-a-cpu.result == 'success' ||
needs.resolve-ci-policy.outputs.bypass_fastfail == 'true')
strategy:
fail-fast: false
max-parallel: ${{ needs.resolve-ci-policy.outputs.cadence == 'weekly' && 1 || 2 }}
matrix:
partition_id: [0, 1]
uses: ./.github/workflows/_run-ci.yml
with:
runs_on: '["h200", "2gpu"]'
container_image: ${{ needs.resolve-ci-image.outputs.container_image }}
ref: ${{ inputs.ref || '' }}
execute_command: >-
python tests/ci/run_suite.py --hw cuda --suite stage-c-2-gpu-h200
--auto-partition-id ${{ matrix.partition_id }} --auto-partition-size 2
--cadence ${{ needs.resolve-ci-policy.outputs.cadence }}
--labels ${{ needs.resolve-ci-policy.outputs.raw_labels }}
${{ github.event_name == 'workflow_dispatch' && '--match-all-labels' || '' }}
secrets: inherit
# The Blackwell fleet is one unpartitioned 8-GPU host: an 8-GPU runner cannot
# coexist with 2/4-GPU runners on the same node, since GitHub would dispatch
# jobs onto overlapping devices. Tests needing fewer than 8 GPUs therefore
# register against this suite too. No partition matrix: one runner serialises
# shards anyway, and the suite's current budget is far inside the job timeout.
stage-c-8-gpu-b200:
needs: [resolve-ci-policy, resolve-ci-image, stage-a-cpu]
if: |
always() && !cancelled() &&
needs.resolve-ci-policy.result == 'success' &&
needs.resolve-ci-image.result == 'success' &&
!contains(fromJSON(needs.resolve-ci-policy.outputs.skipped_stages || '[]'), 'stage-c-8-gpu-b200') &&
(needs.stage-a-cpu.result == 'success' ||
needs.resolve-ci-policy.outputs.bypass_fastfail == 'true')
uses: ./.github/workflows/_run-ci.yml
with:
runs_on: '["b200", "8gpu"]'
container_image: ${{ needs.resolve-ci-image.outputs.container_image }}
ref: ${{ inputs.ref || '' }}
execute_command: >-
python tests/ci/run_suite.py --hw cuda --suite stage-c-8-gpu-b200
--cadence ${{ needs.resolve-ci-policy.outputs.cadence }}
--labels ${{ needs.resolve-ci-policy.outputs.raw_labels }}
${{ github.event_name == 'workflow_dispatch' && '--match-all-labels' || '' }}
secrets: inherit