Skip to content

Integration Tests

Integration Tests #122

name: Integration Tests
on:
workflow_call:
inputs:
gpu-arch:
description: "GPU architecture to test ('cuda' or 'rocm')"
required: false
type: string
default: cuda
execution_mode:
description: >-
Select the jobs for a reusable workflow call: fake_pg runs the 1-GPU
simulated process-group suite, while real_pg runs the 8-GPU real
process-group suite. Direct workflow events ignore this input.
required: false
type: string
default: fake_pg
push:
branches: [ main ]
tags:
- ciflow/fake-pg/*
- ciflow/real-pg/*
paths-ignore:
- 'torchtitan/experiments/**'
schedule:
- cron: '0 */6 * * *'
workflow_dispatch:
inputs:
export_results:
description: Export numerical results instead of comparing with goldens
required: false
type: boolean
default: false
test_name:
description: Run one integration test by name instead of the full suite
required: false
type: string
default: all
concurrency:
group: unit-test-${{ github.workflow }}-${{ github.ref == 'refs/heads/main' && github.run_number || github.ref }}
cancel-in-progress: true
defaults:
run:
shell: bash -l -eo pipefail {0}
permissions:
id-token: write
contents: read
jobs:
# Run Fake PG when CIFlow pushes a ciflow/fake-pg tag for a ready PR.
# Tag pushes are independent of the PR base, so stacked PRs are covered.
# Reusable calls also run this job when execution_mode is fake_pg.
set-matrix-fake-pg:
if: >-
((github.event_name == 'push' &&
startsWith(github.ref, 'refs/tags/ciflow/fake-pg/')) ||
(github.event_name == 'workflow_call' && inputs.execution_mode == 'fake_pg')) &&
(github.repository_owner == 'pytorch' || github.event_name != 'schedule')
uses: ./.github/workflows/set-matrix.yaml
with:
runner-cuda: linux.g5.4xlarge.nvidia.gpu
gpu-arch: ${{ inputs.gpu-arch || 'cuda' }}
# A ciflow/fake-pg tag also runs tests marked real_pg_required so every
# eligible PR gets both tiers, including PRs whose base is another stack PR.
# Main pushes, opt-in ciflow/real-pg tags, schedules, manual runs, and Real PG
# reusable calls run the full Real PG suite.
set-matrix-real-pg:
if: >-
((github.event_name == 'push' &&
(github.ref == 'refs/heads/main' ||
startsWith(github.ref, 'refs/tags/ciflow/fake-pg/') ||
startsWith(github.ref, 'refs/tags/ciflow/real-pg/'))) ||
github.event_name == 'schedule' ||
github.event_name == 'workflow_dispatch' ||
(github.event_name == 'workflow_call' && inputs.execution_mode == 'real_pg')) &&
(github.repository_owner == 'pytorch' || github.event_name != 'schedule')
uses: ./.github/workflows/set-matrix.yaml
with:
gpu-arch: ${{ inputs.gpu-arch || 'cuda' }}
integration-fake-pg:
name: 1 GPU Integration (Fake PG)
needs: set-matrix-fake-pg
if: ${{ needs.set-matrix-fake-pg.outputs.matrix != '' && fromJSON(needs.set-matrix-fake-pg.outputs.matrix).include[0] != null }}
uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main
strategy:
fail-fast: false
matrix: ${{ fromJSON(needs.set-matrix-fake-pg.outputs.matrix) }}
with:
runner: ${{ matrix.runner }}
gpu-arch-type: ${{ matrix.gpu-arch-type }}
gpu-arch-version: ${{ matrix.gpu-arch-version }}
docker-image: ${{ matrix.docker-image }}
repository: pytorch/torchtitan
upload-artifact: integration-1gpu-outputs
timeout: 60
script: |
set -eux
CONDA_ENV=$(conda env list --json | jq -r ".envs | .[-1]")
conda activate "${CONDA_ENV}"
export HF_HOME="$RUNNER_TEMP/hf_home"
export HF_DATASETS_CACHE="$RUNNER_TEMP/hf_home/datasets"
pip config --user set global.progress_bar off
TORCH_SPEC="torch"
if [ -n "${{ matrix.torch-version }}" ]; then
TORCH_SPEC="torch==${{ matrix.torch-version }}"
fi
python -m pip install --force-reinstall --pre \
"${TORCH_SPEC}" torchvision --index-url ${{ matrix.index-url }}
USE_CPP=0 python -m pip install --pre torchao --index-url ${{ matrix.index-url }}
GPU_ARCH="a10g"
if [[ "${{ matrix.gpu-arch-type }}" == "rocm" ]]; then
GPU_ARCH="mi350x"
HIPBLASLT_LIB_DIR="$(python -c 'import os, torch; print(os.path.join(os.path.dirname(torch.__file__), "lib", "hipblaslt", "library"))')"
if [ -d "${HIPBLASLT_LIB_DIR}" ]; then
export HIPBLASLT_TENSILE_LIBPATH="${HIPBLASLT_LIB_DIR}"
fi
fi
sudo mkdir -p "$RUNNER_ARTIFACT_DIR"
sudo mkdir -p "$HF_HOME"
sudo chown -R $(id -u):$(id -g) "$RUNNER_ARTIFACT_DIR"
sudo chown -R $(id -u):$(id -g) "$HF_HOME"
cleanup_artifacts() {
find "$RUNNER_ARTIFACT_DIR" -type d \( -name checkpoint -o -name inference_results \) -prune -exec rm -rf {} +
chmod -R a+rX "$RUNNER_ARTIFACT_DIR"
}
trap cleanup_artifacts EXIT
EXPORT_ARG=""
if [[ "${{ inputs.export_results || false }}" == "true" ]]; then
EXPORT_ARG="--export-numerics"
fi
TEST_NAME_ARG=""
if [[ "${{ inputs.test_name || 'all' }}" != "all" ]]; then
TEST_NAME_ARG="--test_name=${{ inputs.test_name }}"
fi
python -m tests.integration_tests.run_tests \
--gpu_arch_type ${{ matrix.gpu-arch-type }} \
--gpu_arch "$GPU_ARCH" \
--test_suite features,models --execution_mode fake_pg --ngpu 1 \
$EXPORT_ARG $TEST_NAME_ARG \
"$RUNNER_ARTIFACT_DIR"
python -m tests.integration_tests.flux \
--execution_mode fake_pg --ngpu 1 \
$TEST_NAME_ARG \
"$RUNNER_ARTIFACT_DIR/flux"
integration-real-pg:
name: 8 GPU Integration (Real PG - ${{ startsWith(github.ref, 'refs/tags/ciflow/fake-pg/') && 'required subset' || 'full suite' }} - ${{ matrix.test_suite }})
needs: set-matrix-real-pg
if: ${{ (github.event_name != 'workflow_call' || inputs.execution_mode == 'real_pg') && needs.set-matrix-real-pg.outputs.matrix != '' && fromJSON(needs.set-matrix-real-pg.outputs.matrix).include[0] != null }}
uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main
strategy:
fail-fast: false
matrix:
test_suite: [features, models]
runner_config: ${{ fromJSON(needs.set-matrix-real-pg.outputs.matrix).include }}
with:
runner: ${{ matrix.runner_config.runner }}
gpu-arch-type: ${{ matrix.runner_config['gpu-arch-type'] }}
gpu-arch-version: ${{ matrix.runner_config['gpu-arch-version'] }}
docker-image: ${{ matrix.runner_config['docker-image'] }}
repository: pytorch/torchtitan
upload-artifact: integration-8gpu-${{ matrix.test_suite }}-outputs
timeout: 60
script: |
set -eux
CONDA_ENV=$(conda env list --json | jq -r ".envs | .[-1]")
conda activate "${CONDA_ENV}"
export HF_HOME="$RUNNER_TEMP/hf_home"
export HF_DATASETS_CACHE="$RUNNER_TEMP/hf_home/datasets"
pip config --user set global.progress_bar off
TORCH_SPEC="torch"
if [ -n "${{ matrix.runner_config['torch-version'] }}" ]; then
TORCH_SPEC="torch==${{ matrix.runner_config['torch-version'] }}"
fi
python -m pip install --force-reinstall --pre \
"${TORCH_SPEC}" torchvision --index-url ${{ matrix.runner_config['index-url'] }}
USE_CPP=0 python -m pip install --pre torchao --index-url ${{ matrix.runner_config['index-url'] }}
GPU_ARCH="a10g"
if [[ "${{ matrix.runner_config['gpu-arch-type'] }}" == "rocm" ]]; then
GPU_ARCH="mi350x"
HIPBLASLT_LIB_DIR="$(python -c 'import os, torch; print(os.path.join(os.path.dirname(torch.__file__), "lib", "hipblaslt", "library"))')"
if [ -d "${HIPBLASLT_LIB_DIR}" ]; then
export HIPBLASLT_TENSILE_LIBPATH="${HIPBLASLT_LIB_DIR}"
fi
fi
sudo mkdir -p "$RUNNER_ARTIFACT_DIR"
sudo mkdir -p "$HF_HOME"
sudo chown -R $(id -u):$(id -g) "$RUNNER_ARTIFACT_DIR"
sudo chown -R $(id -u):$(id -g) "$HF_HOME"
cleanup_artifacts() {
find "$RUNNER_ARTIFACT_DIR" -type d \( -name checkpoint -o -name inference_results \) -prune -exec rm -rf {} +
chmod -R a+rX "$RUNNER_ARTIFACT_DIR"
}
trap cleanup_artifacts EXIT
EXPORT_ARG=""
if [[ "${{ inputs.export_results || false }}" == "true" ]]; then
EXPORT_ARG="--export-numerics"
fi
TEST_NAME_ARG=""
if [[ "${{ inputs.test_name || 'all' }}" != "all" ]]; then
TEST_NAME_ARG="--test_name=${{ inputs.test_name }}"
fi
# A ciflow/fake-pg tag runs only tests that cannot use Fake PG.
# Main pushes, ciflow/real-pg tags, schedules, manual runs, and Real PG
# workflow calls leave the scope empty and run the full Real PG suite.
TEST_SCOPE_ARG=""
if [[ "$GITHUB_REF" == refs/tags/ciflow/fake-pg/* ]]; then
TEST_SCOPE_ARG="--test_scope=real_pg_required"
fi
python -m tests.integration_tests.run_tests \
--gpu_arch_type ${{ matrix.runner_config['gpu-arch-type'] }} \
--gpu_arch "$GPU_ARCH" \
--test_suite ${{ matrix.test_suite }} --execution_mode real_pg --ngpu 8 \
$EXPORT_ARG $TEST_NAME_ARG $TEST_SCOPE_ARG \
"$RUNNER_ARTIFACT_DIR"
if [[ "${{ matrix.test_suite }}" == "models" ]]; then
python -m tests.integration_tests.flux \
--execution_mode real_pg --ngpu 8 \
$TEST_NAME_ARG $TEST_SCOPE_ARG \
"$RUNNER_ARTIFACT_DIR/flux"
fi