diff --git a/.github/unittest/linux_libs/scripts_gym/run_all.sh b/.github/unittest/linux_libs/scripts_gym/run_all.sh index 4a2c5dd40d4..822bc5a1947 100755 --- a/.github/unittest/linux_libs/scripts_gym/run_all.sh +++ b/.github/unittest/linux_libs/scripts_gym/run_all.sh @@ -206,7 +206,7 @@ run_tests() { test_failed=1 fi - if ! python .github/unittest/helpers/coverage_run_parallel.py -m pytest test/libs --instafail -v --durations 200 -k "gym and not isaac" --mp_fork; then + if ! python .github/unittest/helpers/coverage_run_parallel.py -m pytest test/libs --instafail -v --durations 200 -k "gym and not isaac" --mp_fork_if_no_cuda; then echo "ERROR: test/libs failed for ${version_name}" test_failed=1 fi diff --git a/.github/workflows/benchmarks.yml b/.github/workflows/benchmarks.yml index 2a1abef8aef..377876eb80e 100644 --- a/.github/workflows/benchmarks.yml +++ b/.github/workflows/benchmarks.yml @@ -66,7 +66,7 @@ jobs: if: github.event_name != 'pull_request' needs: validate-report name: ${{ matrix.device }} Pytest benchmark - runs-on: linux.g5.4xlarge.nvidia.gpu + runs-on: mt-l-x86aavx2-11-41-a10g timeout-minutes: 120 strategy: fail-fast: false diff --git a/.github/workflows/benchmarks_pr.yml b/.github/workflows/benchmarks_pr.yml index b29b0b5d70f..a073bb94d70 100644 --- a/.github/workflows/benchmarks_pr.yml +++ b/.github/workflows/benchmarks_pr.yml @@ -71,7 +71,7 @@ jobs: prepare-environment: name: Prepare pinned benchmark environment if: contains(github.event.pull_request.labels.*.name, 'benchmarks/upload') - runs-on: linux.g5.4xlarge.nvidia.gpu + runs-on: mt-l-x86aavx2-11-41-a10g container: image: nvidia/cuda:12.6.3-cudnn-devel-ubuntu22.04@sha256:b3e7fba84d169f46939f00c25be7d016f712a8d651f4756d6a55e693d84d94f2 options: --gpus all --shm-size=8g @@ -153,7 +153,7 @@ jobs: name: ${{ matrix.device }} ${{ matrix.revision }} benchmark if: contains(github.event.pull_request.labels.*.name, 'benchmarks/upload') needs: [prepare-definitions, prepare-environment] - runs-on: linux.g5.4xlarge.nvidia.gpu + runs-on: mt-l-x86aavx2-11-41-a10g strategy: fail-fast: false max-parallel: 4 @@ -315,7 +315,7 @@ jobs: "pr_number": int(os.environ["PR_NUMBER"]), "base_sha": os.environ["BASE_SHA"], "head_sha": os.environ["HEAD_SHA"], - "runner": "linux.g5.4xlarge.nvidia.gpu", + "runner": "mt-l-x86aavx2-11-41-a10g", "image": os.environ["IMAGE"], "python_version": os.environ["PYTHON_VERSION"], "system_environment_sha256": os.environ["SYSTEM_ENVIRONMENT_SHA"], diff --git a/.github/workflows/lint.yml b/.github/workflows/lint.yml index 3eb06b982a1..9ed128e2012 100644 --- a/.github/workflows/lint.yml +++ b/.github/workflows/lint.yml @@ -22,8 +22,9 @@ permissions: jobs: python-source-and-configs: - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: + runner: mt-l-x86iavx512-8-64 repository: pytorch/rl script: | set -euo pipefail @@ -50,8 +51,9 @@ jobs: echo '::endgroup::' c-source: - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: + runner: mt-l-x86iavx512-8-64 repository: pytorch/rl script: | set -euo pipefail diff --git a/.github/workflows/test-linux-examples.yml b/.github/workflows/test-linux-examples.yml index 91fe112c642..166436dc119 100644 --- a/.github/workflows/test-linux-examples.yml +++ b/.github/workflows/test-linux-examples.yml @@ -26,11 +26,11 @@ jobs: cuda_arch_version: ["13.0"] shard: ["1", "2"] fail-fast: false - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.g5.4xlarge.nvidia.gpu + runner: mt-l-x86aavx2-11-41-a10g repository: pytorch/rl - docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04" gpu-arch-type: cuda gpu-arch-version: ${{ matrix.cuda_arch_version }} timeout: 120 diff --git a/.github/workflows/test-linux-habitat.yml b/.github/workflows/test-linux-habitat.yml index 580849525e9..d2914180c71 100644 --- a/.github/workflows/test-linux-habitat.yml +++ b/.github/workflows/test-linux-habitat.yml @@ -28,11 +28,11 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] fail-fast: false - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.g5.12xlarge.nvidia.gpu + runner: mt-l-x86aavx2-45-167-a10g-4 repository: pytorch/rl - docker-image: "nvidia/cuda:12.8.1-cudnn-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.8.1-cudnn-devel-ubuntu24.04" gpu-arch-type: cuda gpu-arch-version: ${{ matrix.cuda_arch_version }} timeout: 90 diff --git a/.github/workflows/test-linux-libs.yml b/.github/workflows/test-linux-libs.yml index a589f557e30..8d563e550b4 100644 --- a/.github/workflows/test-linux-libs.yml +++ b/.github/workflows/test-linux-libs.yml @@ -29,11 +29,11 @@ jobs: # python_version: ["3.10"] # cuda_arch_version: ["12.8"] # if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Data') }} - # uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + # uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main # with: # repository: pytorch/rl - # runner: "linux.g5.4xlarge.nvidia.gpu" - # docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + # runner: "mt-l-x86aavx2-11-41-a10g" + # docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" # timeout: 120 # script: | # if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -63,13 +63,13 @@ jobs: python_version: ["3.11"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/brax') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -99,13 +99,13 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/craftground') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -140,13 +140,13 @@ jobs: python_version: ["3.11"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/mujoco_playground') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -176,13 +176,13 @@ jobs: python_version: ["3.11"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/mjlab') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -212,13 +212,13 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Modules') || contains(github.event.pull_request.labels.*.name, 'Objectives') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -250,11 +250,11 @@ jobs: # python_version: ["3.10"] # cuda_arch_version: ["12.8"] # if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Data') }} - # uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + # uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main # with: # repository: pytorch/rl - # runner: "linux.g5.4xlarge.nvidia.gpu" - # docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + # runner: "mt-l-x86aavx2-11-41-a10g" + # docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" # timeout: 120 # script: | # if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -285,11 +285,11 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/envpool') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + runner: "mt-l-x86aavx2-11-41-a10g" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -322,11 +322,11 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Data') || contains(github.event.pull_request.labels.*.name, 'Data/gendgrl') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + runner: "mt-l-x86aavx2-11-41-a10g" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -357,13 +357,13 @@ jobs: matrix: python_version: ["3.10"] cuda_arch_version: ["12.8"] - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" # gpu-arch-type: "cuda" # gpu-arch-version: "11.6" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -409,13 +409,13 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/jumanji') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -451,13 +451,13 @@ jobs: cuda_arch_version: ["12.8"] fail-fast: false if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/libero') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.8.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.8.1-cudnn-devel-ubuntu24.04" timeout: 120 upload-artifact: test-results-libero script: | @@ -488,14 +488,14 @@ jobs: bash .github/unittest/linux_libs/scripts_libero/post_process.sh unittests-meltingpot: - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/meltingpot') }} with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -530,13 +530,13 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/open_spiel') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -570,13 +570,13 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/chess') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -610,13 +610,13 @@ jobs: python_version: ["3.10.12"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/unity_mlagents') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -650,11 +650,11 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Data') || contains(github.event.pull_request.labels.*.name, 'Data/minari') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + runner: "mt-l-x86aavx2-11-41-a10g" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -682,11 +682,11 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Data') || contains(github.event.pull_request.labels.*.name, 'Data/openx') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + runner: "mt-l-x86aavx2-11-41-a10g" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -715,13 +715,13 @@ jobs: unittests-pettingzoo: if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/pettingzoo') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -756,13 +756,13 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/procgen') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -797,13 +797,13 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/safety_gymnasium') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -838,45 +838,87 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/robohive') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main - with: - repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" - docker-image: "nvidia/cudagl:11.4.0-base" - timeout: 120 - script: | - if [[ "${{ github.ref }}" =~ release/* ]]; then - export RELEASE=1 - export TORCH_VERSION=stable - else - export RELEASE=0 - export TORCH_VERSION=nightly - fi - - set -euo pipefail - export PYTHON_VERSION="3.10" - export CU_VERSION="cu128" - export TAR_OPTIONS="--no-same-owner" - export UPLOAD_CHANNEL="nightly" - export TF_CPP_MIN_LOG_LEVEL=0 - export BATCHED_PIPE_TIMEOUT=60 - export TD_GET_DEFAULTS_TO_NONE=1 - - bash .github/unittest/linux_libs/scripts_robohive/setup_env.sh - bash .github/unittest/linux_libs/scripts_robohive/install_and_run_test.sh - bash .github/unittest/linux_libs/scripts_robohive/post_process.sh - + # Not linux_job_v3: on OSDC the checkout runs inside this container, and + # nvidia/cudagl ships no git, so it would degrade to a REST tarball with no + # .git. The image is kept as-is -- it is the only one providing the EGL + # stack this suite renders with, and setup_env.sh installs no GL itself. + runs-on: mt-l-x86aavx2-11-41-a10g + container: + image: "nvidia/cudagl:11.4.0-base" + timeout-minutes: 120 + env: + PR_NUMBER: ${{ github.event.pull_request.number }} + REPOSITORY: pytorch/rl + steps: + - name: Install git + shell: bash + run: | + set -eux + apt-get update + apt-get install -y --no-install-recommends git + + - name: Check out pytorch/rl + uses: actions/checkout@v4 + with: + path: pytorch/rl + + - name: Set up runner directories + shell: bash + run: | + set -euo pipefail + for name in ARTIFACT:artifacts TEST_RESULTS:test-results DOCS:docs; do + var="RUNNER_${name%%:*}_DIR" + dir="${RUNNER_TEMP}/${name##*:}" + rm -rf "${dir}" + mkdir -p "${dir}" + echo "${var}=${dir}" >> "${GITHUB_ENV}" + done + + - name: Run script + shell: bash + working-directory: pytorch/rl + run: | + set -eou pipefail + eval "$(conda shell.bash hook)" + set -x + if [[ "${{ github.ref }}" =~ release/* ]]; then + export RELEASE=1 + export TORCH_VERSION=stable + else + export RELEASE=0 + export TORCH_VERSION=nightly + fi + + set -euo pipefail + export PYTHON_VERSION="3.10" + export CU_VERSION="cu128" + export TAR_OPTIONS="--no-same-owner" + export UPLOAD_CHANNEL="nightly" + export TF_CPP_MIN_LOG_LEVEL=0 + export BATCHED_PIPE_TIMEOUT=60 + export TD_GET_DEFAULTS_TO_NONE=1 + + bash .github/unittest/linux_libs/scripts_robohive/setup_env.sh + bash .github/unittest/linux_libs/scripts_robohive/install_and_run_test.sh + bash .github/unittest/linux_libs/scripts_robohive/post_process.sh + + - name: Surface failing tests + if: always() + uses: pmeier/pytest-results-action@a2c1430e2bddadbad9f49a6f9b879f062c6b19b1 + with: + path: ${{ env.RUNNER_TEST_RESULTS_DIR }} + fail-on-empty: false unittests-roboset: strategy: matrix: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Data') || contains(github.event.pull_request.labels.*.name, 'Data/roboset') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + runner: "mt-l-x86aavx2-11-41-a10g" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -908,13 +950,13 @@ jobs: matrix: python_version: ["3.10"] cuda_arch_version: ["12.8"] - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" # gpu-arch-type: cuda # gpu-arch-version: "11.7" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -947,13 +989,13 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/smacv2') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -988,11 +1030,11 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Data') || contains(github.event.pull_request.labels.*.name, 'Data/vd4rl') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + runner: "mt-l-x86aavx2-11-41-a10g" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -1025,13 +1067,13 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/vmas') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -1066,13 +1108,13 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Integrations') || contains(github.event.pull_request.labels.*.name, 'Integrations/torch_geometric') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then diff --git a/.github/workflows/test-linux-llm.yml b/.github/workflows/test-linux-llm.yml index 4ec7b831b4c..30f0b1b63de 100644 --- a/.github/workflows/test-linux-llm.yml +++ b/.github/workflows/test-linux-llm.yml @@ -29,11 +29,11 @@ jobs: matrix: python_version: ["3.12"] cuda_arch_version: ["12.9"] - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g6.4xlarge.experimental.nvidia.gpu" - docker-image: "pytorch/pytorch:2.8.0-cuda12.9-cudnn9-devel" + runner: "mt-l-x86aavx2-11-41-l4" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.9.2-cudnn-devel-ubuntu24.04" timeout: 60 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -72,11 +72,11 @@ jobs: matrix: python_version: ["3.12"] cuda_arch_version: ["12.9"] - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g6.4xlarge.experimental.nvidia.gpu" - docker-image: "pytorch/pytorch:2.8.0-cuda12.9-cudnn9-devel" + runner: "mt-l-x86aavx2-11-41-l4" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.9.2-cudnn-devel-ubuntu24.04" timeout: 60 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then diff --git a/.github/workflows/test-linux-mujoco.yml b/.github/workflows/test-linux-mujoco.yml index 55df6a07cf7..e7e78504f2b 100644 --- a/.github/workflows/test-linux-mujoco.yml +++ b/.github/workflows/test-linux-mujoco.yml @@ -37,13 +37,13 @@ jobs: matrix: python_version: ["3.11"] cuda_arch_version: ["12.8"] - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.8.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.8.1-cudnn-devel-ubuntu24.04" timeout: 90 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then diff --git a/.github/workflows/test-linux-sota.yml b/.github/workflows/test-linux-sota.yml index 64a2c512b0a..ae051475723 100644 --- a/.github/workflows/test-linux-sota.yml +++ b/.github/workflows/test-linux-sota.yml @@ -33,11 +33,11 @@ jobs: # filtering in .github/unittest/linux_sota/scripts/test_sota.py. shard: ["1", "2"] fail-fast: false - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.g5.4xlarge.nvidia.gpu + runner: mt-l-x86aavx2-11-41-a10g repository: pytorch/rl - docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04" gpu-arch-type: cuda gpu-arch-version: ${{ matrix.cuda_arch_version }} timeout: 90 diff --git a/.github/workflows/test-linux-tutorials.yml b/.github/workflows/test-linux-tutorials.yml index 41d5152cbe1..459e8455aa4 100644 --- a/.github/workflows/test-linux-tutorials.yml +++ b/.github/workflows/test-linux-tutorials.yml @@ -27,11 +27,11 @@ jobs: cuda_arch_version: ["13.0"] fail-fast: false # Run on all PRs and pushes to main/nightly/release branches - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.g5.4xlarge.nvidia.gpu + runner: mt-l-x86aavx2-11-41-a10g repository: pytorch/rl - docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04" gpu-arch-type: cuda gpu-arch-version: ${{ matrix.cuda_arch_version }} timeout: 120 diff --git a/.github/workflows/test-linux.yml b/.github/workflows/test-linux.yml index 5de7746380b..f6f6579e2dc 100644 --- a/.github/workflows/test-linux.yml +++ b/.github/workflows/test-linux.yml @@ -29,11 +29,11 @@ jobs: matrix: python_version: ["3.10", "3.14"] fail-fast: false - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.4xlarge + runner: mt-l-x86iavx512-16-128 repository: pytorch/rl - docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04" timeout: 90 script: | set -euo pipefail @@ -53,11 +53,11 @@ jobs: python_version: ["3.10", "3.13", "3.14"] shard: ["bulk", "collectors", "mp"] fail-fast: false - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.12xlarge + runner: mt-l-x86iavx512-48-384 repository: pytorch/rl - docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04" timeout: 120 upload-artifact: test-results-cpu-${{ matrix.python_version }}-shard-${{ matrix.shard }} script: | @@ -91,11 +91,11 @@ jobs: python_version: ["3.11", "3.12"] shard: ["bulk", "collectors", "mp"] fail-fast: false - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.12xlarge + runner: mt-l-x86iavx512-48-384 repository: pytorch/rl - docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04" timeout: 120 upload-artifact: test-results-cpu-${{ matrix.python_version }}-shard-${{ matrix.shard }} script: | @@ -129,11 +129,11 @@ jobs: # Shard 3: all other tests shard: ["1", "2", "3"] fail-fast: false - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.g5.4xlarge.nvidia.gpu + runner: mt-l-x86aavx2-11-41-a10g repository: pytorch/rl - docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04" gpu-arch-type: cuda gpu-arch-version: ${{ matrix.cuda_arch_version }} timeout: 120 @@ -173,11 +173,11 @@ jobs: python_version: ["3.12"] cuda_arch_version: ["13.0"] fail-fast: false - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.g5.4xlarge.nvidia.gpu + runner: mt-l-x86aavx2-11-41-a10g repository: pytorch/rl - docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04" gpu-arch-type: cuda gpu-arch-version: ${{ matrix.cuda_arch_version }} timeout: 120 @@ -222,11 +222,11 @@ jobs: # serial shards; see linux_olddeps/scripts_gym_0_13/run_test.sh. shard: ["transforms", "quarantine", "remainder"] fail-fast: false - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.g5.4xlarge.nvidia.gpu + runner: mt-l-x86aavx2-11-41-a10g repository: pytorch/rl - docker-image: "nvidia/cuda:11.8.0-cudnn8-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda11.8.0-cudnn8-devel-ubuntu22.04" gpu-arch-type: cuda gpu-arch-version: ${{ matrix.cuda_arch_version }} timeout: 120 @@ -265,11 +265,11 @@ jobs: python_version: ["3.12"] cuda_arch_version: ["13.0"] fail-fast: false - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.g5.4xlarge.nvidia.gpu + runner: mt-l-x86aavx2-11-41-a10g repository: pytorch/rl - docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04" gpu-arch-type: cuda gpu-arch-version: ${{ matrix.cuda_arch_version }} timeout: 120 @@ -306,11 +306,11 @@ jobs: python_version: ["3.12"] cuda_arch_version: ["13.0"] fail-fast: false - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.g5.4xlarge.nvidia.gpu + runner: mt-l-x86aavx2-11-41-a10g repository: pytorch/rl - docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04" gpu-arch-type: cuda gpu-arch-version: ${{ matrix.cuda_arch_version }} timeout: 60 @@ -343,11 +343,11 @@ jobs: # Test sharding: split tests into 3 parallel jobs for faster execution shard: ["1", "2", "3"] fail-fast: false - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.g5.4xlarge.nvidia.gpu + runner: mt-l-x86aavx2-11-41-a10g repository: pytorch/rl - docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04" gpu-arch-type: cuda gpu-arch-version: ${{ matrix.cuda_arch_version }} timeout: 120 @@ -389,11 +389,11 @@ jobs: python_version: ["3.12"] # "3.9", "3.10", "3.11" cuda_arch_version: ["13.0"] # "11.6", "11.7" fail-fast: false - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.g5.4xlarge.nvidia.gpu + runner: mt-l-x86aavx2-11-41-a10g repository: pytorch/rl - docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04" gpu-arch-type: cuda gpu-arch-version: ${{ matrix.cuda_arch_version }} timeout: 120