Integration Tests #112
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| name: Integration Tests | |
| on: | |
| workflow_call: | |
| inputs: | |
| gpu-arch: | |
| description: "GPU architecture to test ('cuda' or 'rocm')" | |
| required: false | |
| type: string | |
| default: cuda | |
| execution_mode: | |
| description: "Process-group mode for reusable workflow calls" | |
| required: false | |
| type: string | |
| default: fake_pg | |
| push: | |
| branches: [ main ] | |
| tags: | |
| - ciflow/8gpu/* | |
| paths-ignore: | |
| - 'torchtitan/experiments/**' | |
| pull_request: | |
| types: [opened, synchronize, reopened, ready_for_review] | |
| branches: [ main ] | |
| paths-ignore: | |
| - 'torchtitan/experiments/**' | |
| schedule: | |
| - cron: '0 */6 * * *' | |
| workflow_dispatch: | |
| inputs: | |
| export_results: | |
| description: Export numerical results instead of comparing with goldens | |
| required: false | |
| type: boolean | |
| default: false | |
| test_name: | |
| description: Run one integration test by name instead of the full suite | |
| required: false | |
| type: string | |
| default: all | |
| concurrency: | |
| group: unit-test-${{ github.workflow }}-${{ github.ref == 'refs/heads/main' && github.run_number || github.ref }} | |
| cancel-in-progress: true | |
| defaults: | |
| run: | |
| shell: bash -l -eo pipefail {0} | |
| permissions: | |
| id-token: write | |
| contents: read | |
| jobs: | |
| set-matrix-fake-pg: | |
| if: >- | |
| (github.event_name == 'pull_request' || | |
| (github.event_name == 'workflow_call' && inputs.execution_mode == 'fake_pg')) && | |
| (github.repository_owner == 'pytorch' || github.event_name != 'schedule') | |
| uses: ./.github/workflows/set-matrix.yaml | |
| with: | |
| runner-cuda: linux.g5.4xlarge.nvidia.gpu | |
| gpu-arch: ${{ inputs.gpu-arch || 'cuda' }} | |
| set-matrix-real-pg: | |
| if: >- | |
| (github.event_name != 'workflow_call' || inputs.execution_mode == 'real_pg') && | |
| (github.repository_owner == 'pytorch' || github.event_name != 'schedule') | |
| uses: ./.github/workflows/set-matrix.yaml | |
| with: | |
| gpu-arch: ${{ inputs.gpu-arch || 'cuda' }} | |
| integration-fake-pg: | |
| name: 1 GPU Integration (Fake PG) | |
| needs: set-matrix-fake-pg | |
| if: ${{ (github.event_name == 'pull_request' || (github.event_name == 'workflow_call' && inputs.execution_mode == 'fake_pg')) && needs.set-matrix-fake-pg.outputs.matrix != '' && fromJSON(needs.set-matrix-fake-pg.outputs.matrix).include[0] != null }} | |
| uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main | |
| strategy: | |
| fail-fast: false | |
| matrix: ${{ fromJSON(needs.set-matrix-fake-pg.outputs.matrix) }} | |
| with: | |
| runner: ${{ matrix.runner }} | |
| gpu-arch-type: ${{ matrix.gpu-arch-type }} | |
| gpu-arch-version: ${{ matrix.gpu-arch-version }} | |
| docker-image: ${{ matrix.docker-image }} | |
| repository: pytorch/torchtitan | |
| upload-artifact: integration-1gpu-outputs | |
| timeout: 60 | |
| script: | | |
| set -eux | |
| CONDA_ENV=$(conda env list --json | jq -r ".envs | .[-1]") | |
| conda activate "${CONDA_ENV}" | |
| export HF_HOME="$RUNNER_TEMP/hf_home" | |
| export HF_DATASETS_CACHE="$RUNNER_TEMP/hf_home/datasets" | |
| pip config --user set global.progress_bar off | |
| TORCH_SPEC="torch" | |
| if [ -n "${{ matrix.torch-version }}" ]; then | |
| TORCH_SPEC="torch==${{ matrix.torch-version }}" | |
| fi | |
| python -m pip install --force-reinstall --pre \ | |
| "${TORCH_SPEC}" torchvision --index-url ${{ matrix.index-url }} | |
| USE_CPP=0 python -m pip install --pre torchao --index-url ${{ matrix.index-url }} | |
| if [[ "${{ matrix.gpu-arch-type }}" == "rocm" ]]; then | |
| export HIPBLASLT_TENSILE_LIBPATH="$(python -c 'import os, torch; print(os.path.join(os.path.dirname(torch.__file__), "lib", "hipblaslt", "library"))')" | |
| fi | |
| sudo mkdir -p "$RUNNER_ARTIFACT_DIR" | |
| sudo mkdir -p "$HF_HOME" | |
| sudo chown -R $(id -u):$(id -g) "$RUNNER_ARTIFACT_DIR" | |
| sudo chown -R $(id -u):$(id -g) "$HF_HOME" | |
| cleanup_artifacts() { | |
| find "$RUNNER_ARTIFACT_DIR" -type d \( -name checkpoint -o -name inference_results \) -prune -exec rm -rf {} + | |
| chmod -R a+rX "$RUNNER_ARTIFACT_DIR" | |
| } | |
| trap cleanup_artifacts EXIT | |
| EXPORT_ARG="" | |
| if [[ "${{ inputs.export_results || false }}" == "true" ]]; then | |
| EXPORT_ARG="--export-numerics" | |
| fi | |
| TEST_NAME_ARG="" | |
| if [[ "${{ inputs.test_name || 'all' }}" != "all" ]]; then | |
| TEST_NAME_ARG="--test_name=${{ inputs.test_name }}" | |
| fi | |
| python -m tests.integration_tests.run_tests \ | |
| --gpu_arch_type ${{ matrix.gpu-arch-type }} \ | |
| --test_suite features,models --execution_mode fake_pg --ngpu 1 \ | |
| $EXPORT_ARG $TEST_NAME_ARG \ | |
| "$RUNNER_ARTIFACT_DIR" | |
| python -m tests.integration_tests.flux \ | |
| --execution_mode fake_pg --ngpu 1 \ | |
| $TEST_NAME_ARG \ | |
| "$RUNNER_ARTIFACT_DIR/flux" | |
| integration-real-pg: | |
| name: 8 GPU Integration (Real PG - ${{ github.event_name == 'pull_request' && 'required subset' || 'full suite' }} - ${{ matrix.test_suite }}) | |
| needs: set-matrix-real-pg | |
| if: ${{ (github.event_name != 'workflow_call' || inputs.execution_mode == 'real_pg') && needs.set-matrix-real-pg.outputs.matrix != '' && fromJSON(needs.set-matrix-real-pg.outputs.matrix).include[0] != null }} | |
| uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main | |
| strategy: | |
| fail-fast: false | |
| matrix: | |
| test_suite: [features, models] | |
| runner_config: ${{ fromJSON(needs.set-matrix-real-pg.outputs.matrix).include }} | |
| with: | |
| runner: ${{ matrix.runner_config.runner }} | |
| gpu-arch-type: ${{ matrix.runner_config['gpu-arch-type'] }} | |
| gpu-arch-version: ${{ matrix.runner_config['gpu-arch-version'] }} | |
| docker-image: ${{ matrix.runner_config['docker-image'] }} | |
| repository: pytorch/torchtitan | |
| upload-artifact: integration-8gpu-${{ matrix.test_suite }}-outputs | |
| timeout: 60 | |
| script: | | |
| set -eux | |
| CONDA_ENV=$(conda env list --json | jq -r ".envs | .[-1]") | |
| conda activate "${CONDA_ENV}" | |
| export HF_HOME="$RUNNER_TEMP/hf_home" | |
| export HF_DATASETS_CACHE="$RUNNER_TEMP/hf_home/datasets" | |
| pip config --user set global.progress_bar off | |
| TORCH_SPEC="torch" | |
| if [ -n "${{ matrix.runner_config['torch-version'] }}" ]; then | |
| TORCH_SPEC="torch==${{ matrix.runner_config['torch-version'] }}" | |
| fi | |
| python -m pip install --force-reinstall --pre \ | |
| "${TORCH_SPEC}" torchvision --index-url ${{ matrix.runner_config['index-url'] }} | |
| USE_CPP=0 python -m pip install --pre torchao --index-url ${{ matrix.runner_config['index-url'] }} | |
| if [[ "${{ matrix.runner_config['gpu-arch-type'] }}" == "rocm" ]]; then | |
| export HIPBLASLT_TENSILE_LIBPATH="$(python -c 'import os, torch; print(os.path.join(os.path.dirname(torch.__file__), "lib", "hipblaslt", "library"))')" | |
| fi | |
| sudo mkdir -p "$RUNNER_ARTIFACT_DIR" | |
| sudo mkdir -p "$HF_HOME" | |
| sudo chown -R $(id -u):$(id -g) "$RUNNER_ARTIFACT_DIR" | |
| sudo chown -R $(id -u):$(id -g) "$HF_HOME" | |
| cleanup_artifacts() { | |
| find "$RUNNER_ARTIFACT_DIR" -type d \( -name checkpoint -o -name inference_results \) -prune -exec rm -rf {} + | |
| chmod -R a+rX "$RUNNER_ARTIFACT_DIR" | |
| } | |
| trap cleanup_artifacts EXIT | |
| EXPORT_ARG="" | |
| if [[ "${{ inputs.export_results || false }}" == "true" ]]; then | |
| EXPORT_ARG="--export-numerics" | |
| fi | |
| TEST_NAME_ARG="" | |
| if [[ "${{ inputs.test_name || 'all' }}" != "all" ]]; then | |
| TEST_NAME_ARG="--test_name=${{ inputs.test_name }}" | |
| fi | |
| TEST_SCOPE_ARG="" | |
| if [[ "${{ github.event_name }}" == "pull_request" ]]; then | |
| TEST_SCOPE_ARG="--test_scope=real_pg_required" | |
| fi | |
| python -m tests.integration_tests.run_tests \ | |
| --gpu_arch_type ${{ matrix.runner_config['gpu-arch-type'] }} \ | |
| --test_suite ${{ matrix.test_suite }} --execution_mode real_pg --ngpu 8 \ | |
| $EXPORT_ARG $TEST_NAME_ARG $TEST_SCOPE_ARG \ | |
| "$RUNNER_ARTIFACT_DIR" | |
| if [[ "${{ matrix.test_suite }}" == "models" ]]; then | |
| python -m tests.integration_tests.flux \ | |
| --execution_mode real_pg --ngpu 8 \ | |
| $TEST_NAME_ARG $TEST_SCOPE_ARG \ | |
| "$RUNNER_ARTIFACT_DIR/flux" | |
| fi |