From a6421f08a8fbc04da8e3ff41e6cd2d2956cfc97b Mon Sep 17 00:00:00 2001 From: Huy Do Date: Wed, 7 Oct 2026 19:14:17 -0700 Subject: [PATCH 1/6] [Refactor] Move the Linux CI jobs to OSDC runners Moves the Linux test and lint jobs across 11 workflows onto OSDC pods. | before | after | |---|---| | `linux.g5.4xlarge.nvidia.gpu` | `mt-l-x86aavx2-11-41-a10g` | | `linux.g6.4xlarge.experimental.nvidia.gpu` | `mt-l-x86aavx2-11-41-l4` | | `linux.12xlarge` | `mt-l-x86iavx512-48-384` | | `linux.4xlarge` | `mt-l-x86iavx512-16-128` | Each pod sits on the same instance the label it replaces resolved to, so GPU model and count are unchanged and only the Kubernetes overhead comes off. `linux_job_v2` -> `v3` comes with it, per job: OSDC rejects a job with no container. The lint jobs set no `runner:` at all, so they were quietly taking v2's `linux.2xlarge` default. They now pin an OSDC pod explicitly. On OSDC the checkout runs inside the job container and no stock nvidia/cuda image ships git, so 42 jobs move to the images test-infra publishes for this -- the same bases plus git: nvidia/cuda:12.4.0-devel-ubuntu22.04 -> osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04 nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04 -> osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04 nvidia/cuda:12.8.x-*-ubuntu22.04 -> osdc-cuda:cuda12.8.1-cudnn-devel-ubuntu24.04 nvidia/cuda:11.8.0-cudnn8-devel-ubuntu22.04 -> osdc-cuda:cuda11.8.0-cudnn8-devel-ubuntu22.04 pytorch/pytorch:2.8.0-cuda12.9-cudnn9-devel -> osdc-cuda:cuda12.9.2-cudnn-devel-ubuntu24.04 unittests-robohive keeps nvidia/cudagl:11.4.0-base -- it is the only image providing the EGL stack that suite renders with -- so it runs standalone with an explicit git install instead of through v3. test-linux-libs.yml's `unittests-isaaclab` stays on EC2: it calls a fork's reusable workflow that cannot be confirmed to set a container. docs.yml is handled in #4524. Wheel builds are in #4487. Authored with Claude Code. ghstack-source-id: b70316e68ea8da6cb6688cbc571d5611600fe1a2 Pull-Request: https://github.com/pytorch/rl/pull/4527 --- .github/workflows/benchmarks.yml | 2 +- .github/workflows/benchmarks_pr.yml | 6 +- .github/workflows/lint.yml | 6 +- .github/workflows/test-linux-examples.yml | 6 +- .github/workflows/test-linux-habitat.yml | 6 +- .github/workflows/test-linux-libs.yml | 260 ++++++++++++--------- .github/workflows/test-linux-llm.yml | 12 +- .github/workflows/test-linux-mujoco.yml | 6 +- .github/workflows/test-linux-sota.yml | 6 +- .github/workflows/test-linux-tutorials.yml | 6 +- .github/workflows/test-linux.yml | 60 ++--- 11 files changed, 210 insertions(+), 166 deletions(-) diff --git a/.github/workflows/benchmarks.yml b/.github/workflows/benchmarks.yml index 2a1abef8aef..377876eb80e 100644 --- a/.github/workflows/benchmarks.yml +++ b/.github/workflows/benchmarks.yml @@ -66,7 +66,7 @@ jobs: if: github.event_name != 'pull_request' needs: validate-report name: ${{ matrix.device }} Pytest benchmark - runs-on: linux.g5.4xlarge.nvidia.gpu + runs-on: mt-l-x86aavx2-11-41-a10g timeout-minutes: 120 strategy: fail-fast: false diff --git a/.github/workflows/benchmarks_pr.yml b/.github/workflows/benchmarks_pr.yml index b29b0b5d70f..a073bb94d70 100644 --- a/.github/workflows/benchmarks_pr.yml +++ b/.github/workflows/benchmarks_pr.yml @@ -71,7 +71,7 @@ jobs: prepare-environment: name: Prepare pinned benchmark environment if: contains(github.event.pull_request.labels.*.name, 'benchmarks/upload') - runs-on: linux.g5.4xlarge.nvidia.gpu + runs-on: mt-l-x86aavx2-11-41-a10g container: image: nvidia/cuda:12.6.3-cudnn-devel-ubuntu22.04@sha256:b3e7fba84d169f46939f00c25be7d016f712a8d651f4756d6a55e693d84d94f2 options: --gpus all --shm-size=8g @@ -153,7 +153,7 @@ jobs: name: ${{ matrix.device }} ${{ matrix.revision }} benchmark if: contains(github.event.pull_request.labels.*.name, 'benchmarks/upload') needs: [prepare-definitions, prepare-environment] - runs-on: linux.g5.4xlarge.nvidia.gpu + runs-on: mt-l-x86aavx2-11-41-a10g strategy: fail-fast: false max-parallel: 4 @@ -315,7 +315,7 @@ jobs: "pr_number": int(os.environ["PR_NUMBER"]), "base_sha": os.environ["BASE_SHA"], "head_sha": os.environ["HEAD_SHA"], - "runner": "linux.g5.4xlarge.nvidia.gpu", + "runner": "mt-l-x86aavx2-11-41-a10g", "image": os.environ["IMAGE"], "python_version": os.environ["PYTHON_VERSION"], "system_environment_sha256": os.environ["SYSTEM_ENVIRONMENT_SHA"], diff --git a/.github/workflows/lint.yml b/.github/workflows/lint.yml index 3eb06b982a1..9ed128e2012 100644 --- a/.github/workflows/lint.yml +++ b/.github/workflows/lint.yml @@ -22,8 +22,9 @@ permissions: jobs: python-source-and-configs: - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: + runner: mt-l-x86iavx512-8-64 repository: pytorch/rl script: | set -euo pipefail @@ -50,8 +51,9 @@ jobs: echo '::endgroup::' c-source: - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: + runner: mt-l-x86iavx512-8-64 repository: pytorch/rl script: | set -euo pipefail diff --git a/.github/workflows/test-linux-examples.yml b/.github/workflows/test-linux-examples.yml index 91fe112c642..166436dc119 100644 --- a/.github/workflows/test-linux-examples.yml +++ b/.github/workflows/test-linux-examples.yml @@ -26,11 +26,11 @@ jobs: cuda_arch_version: ["13.0"] shard: ["1", "2"] fail-fast: false - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.g5.4xlarge.nvidia.gpu + runner: mt-l-x86aavx2-11-41-a10g repository: pytorch/rl - docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04" gpu-arch-type: cuda gpu-arch-version: ${{ matrix.cuda_arch_version }} timeout: 120 diff --git a/.github/workflows/test-linux-habitat.yml b/.github/workflows/test-linux-habitat.yml index 580849525e9..d2914180c71 100644 --- a/.github/workflows/test-linux-habitat.yml +++ b/.github/workflows/test-linux-habitat.yml @@ -28,11 +28,11 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] fail-fast: false - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.g5.12xlarge.nvidia.gpu + runner: mt-l-x86aavx2-45-167-a10g-4 repository: pytorch/rl - docker-image: "nvidia/cuda:12.8.1-cudnn-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.8.1-cudnn-devel-ubuntu24.04" gpu-arch-type: cuda gpu-arch-version: ${{ matrix.cuda_arch_version }} timeout: 90 diff --git a/.github/workflows/test-linux-libs.yml b/.github/workflows/test-linux-libs.yml index a589f557e30..8d563e550b4 100644 --- a/.github/workflows/test-linux-libs.yml +++ b/.github/workflows/test-linux-libs.yml @@ -29,11 +29,11 @@ jobs: # python_version: ["3.10"] # cuda_arch_version: ["12.8"] # if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Data') }} - # uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + # uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main # with: # repository: pytorch/rl - # runner: "linux.g5.4xlarge.nvidia.gpu" - # docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + # runner: "mt-l-x86aavx2-11-41-a10g" + # docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" # timeout: 120 # script: | # if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -63,13 +63,13 @@ jobs: python_version: ["3.11"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/brax') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -99,13 +99,13 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/craftground') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -140,13 +140,13 @@ jobs: python_version: ["3.11"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/mujoco_playground') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -176,13 +176,13 @@ jobs: python_version: ["3.11"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/mjlab') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -212,13 +212,13 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Modules') || contains(github.event.pull_request.labels.*.name, 'Objectives') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -250,11 +250,11 @@ jobs: # python_version: ["3.10"] # cuda_arch_version: ["12.8"] # if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Data') }} - # uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + # uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main # with: # repository: pytorch/rl - # runner: "linux.g5.4xlarge.nvidia.gpu" - # docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + # runner: "mt-l-x86aavx2-11-41-a10g" + # docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" # timeout: 120 # script: | # if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -285,11 +285,11 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/envpool') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + runner: "mt-l-x86aavx2-11-41-a10g" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -322,11 +322,11 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Data') || contains(github.event.pull_request.labels.*.name, 'Data/gendgrl') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + runner: "mt-l-x86aavx2-11-41-a10g" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -357,13 +357,13 @@ jobs: matrix: python_version: ["3.10"] cuda_arch_version: ["12.8"] - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" # gpu-arch-type: "cuda" # gpu-arch-version: "11.6" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -409,13 +409,13 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/jumanji') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -451,13 +451,13 @@ jobs: cuda_arch_version: ["12.8"] fail-fast: false if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/libero') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.8.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.8.1-cudnn-devel-ubuntu24.04" timeout: 120 upload-artifact: test-results-libero script: | @@ -488,14 +488,14 @@ jobs: bash .github/unittest/linux_libs/scripts_libero/post_process.sh unittests-meltingpot: - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/meltingpot') }} with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -530,13 +530,13 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/open_spiel') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -570,13 +570,13 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/chess') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -610,13 +610,13 @@ jobs: python_version: ["3.10.12"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/unity_mlagents') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -650,11 +650,11 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Data') || contains(github.event.pull_request.labels.*.name, 'Data/minari') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + runner: "mt-l-x86aavx2-11-41-a10g" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -682,11 +682,11 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Data') || contains(github.event.pull_request.labels.*.name, 'Data/openx') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + runner: "mt-l-x86aavx2-11-41-a10g" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -715,13 +715,13 @@ jobs: unittests-pettingzoo: if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/pettingzoo') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -756,13 +756,13 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/procgen') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -797,13 +797,13 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/safety_gymnasium') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -838,45 +838,87 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/robohive') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main - with: - repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" - docker-image: "nvidia/cudagl:11.4.0-base" - timeout: 120 - script: | - if [[ "${{ github.ref }}" =~ release/* ]]; then - export RELEASE=1 - export TORCH_VERSION=stable - else - export RELEASE=0 - export TORCH_VERSION=nightly - fi - - set -euo pipefail - export PYTHON_VERSION="3.10" - export CU_VERSION="cu128" - export TAR_OPTIONS="--no-same-owner" - export UPLOAD_CHANNEL="nightly" - export TF_CPP_MIN_LOG_LEVEL=0 - export BATCHED_PIPE_TIMEOUT=60 - export TD_GET_DEFAULTS_TO_NONE=1 - - bash .github/unittest/linux_libs/scripts_robohive/setup_env.sh - bash .github/unittest/linux_libs/scripts_robohive/install_and_run_test.sh - bash .github/unittest/linux_libs/scripts_robohive/post_process.sh - + # Not linux_job_v3: on OSDC the checkout runs inside this container, and + # nvidia/cudagl ships no git, so it would degrade to a REST tarball with no + # .git. The image is kept as-is -- it is the only one providing the EGL + # stack this suite renders with, and setup_env.sh installs no GL itself. + runs-on: mt-l-x86aavx2-11-41-a10g + container: + image: "nvidia/cudagl:11.4.0-base" + timeout-minutes: 120 + env: + PR_NUMBER: ${{ github.event.pull_request.number }} + REPOSITORY: pytorch/rl + steps: + - name: Install git + shell: bash + run: | + set -eux + apt-get update + apt-get install -y --no-install-recommends git + + - name: Check out pytorch/rl + uses: actions/checkout@v4 + with: + path: pytorch/rl + + - name: Set up runner directories + shell: bash + run: | + set -euo pipefail + for name in ARTIFACT:artifacts TEST_RESULTS:test-results DOCS:docs; do + var="RUNNER_${name%%:*}_DIR" + dir="${RUNNER_TEMP}/${name##*:}" + rm -rf "${dir}" + mkdir -p "${dir}" + echo "${var}=${dir}" >> "${GITHUB_ENV}" + done + + - name: Run script + shell: bash + working-directory: pytorch/rl + run: | + set -eou pipefail + eval "$(conda shell.bash hook)" + set -x + if [[ "${{ github.ref }}" =~ release/* ]]; then + export RELEASE=1 + export TORCH_VERSION=stable + else + export RELEASE=0 + export TORCH_VERSION=nightly + fi + + set -euo pipefail + export PYTHON_VERSION="3.10" + export CU_VERSION="cu128" + export TAR_OPTIONS="--no-same-owner" + export UPLOAD_CHANNEL="nightly" + export TF_CPP_MIN_LOG_LEVEL=0 + export BATCHED_PIPE_TIMEOUT=60 + export TD_GET_DEFAULTS_TO_NONE=1 + + bash .github/unittest/linux_libs/scripts_robohive/setup_env.sh + bash .github/unittest/linux_libs/scripts_robohive/install_and_run_test.sh + bash .github/unittest/linux_libs/scripts_robohive/post_process.sh + + - name: Surface failing tests + if: always() + uses: pmeier/pytest-results-action@a2c1430e2bddadbad9f49a6f9b879f062c6b19b1 + with: + path: ${{ env.RUNNER_TEST_RESULTS_DIR }} + fail-on-empty: false unittests-roboset: strategy: matrix: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Data') || contains(github.event.pull_request.labels.*.name, 'Data/roboset') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + runner: "mt-l-x86aavx2-11-41-a10g" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -908,13 +950,13 @@ jobs: matrix: python_version: ["3.10"] cuda_arch_version: ["12.8"] - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" # gpu-arch-type: cuda # gpu-arch-version: "11.7" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -947,13 +989,13 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/smacv2') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -988,11 +1030,11 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Data') || contains(github.event.pull_request.labels.*.name, 'Data/vd4rl') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + runner: "mt-l-x86aavx2-11-41-a10g" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -1025,13 +1067,13 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Environments') || contains(github.event.pull_request.labels.*.name, 'Environments/vmas') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -1066,13 +1108,13 @@ jobs: python_version: ["3.10"] cuda_arch_version: ["12.8"] if: ${{ github.event_name == 'push' || github.event_name == 'workflow_call' || github.event_name == 'workflow_dispatch' || contains(github.event.pull_request.labels.*.name, 'Integrations') || contains(github.event.pull_request.labels.*.name, 'Integrations/torch_geometric') }} - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.4.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" timeout: 120 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then diff --git a/.github/workflows/test-linux-llm.yml b/.github/workflows/test-linux-llm.yml index 4ec7b831b4c..30f0b1b63de 100644 --- a/.github/workflows/test-linux-llm.yml +++ b/.github/workflows/test-linux-llm.yml @@ -29,11 +29,11 @@ jobs: matrix: python_version: ["3.12"] cuda_arch_version: ["12.9"] - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g6.4xlarge.experimental.nvidia.gpu" - docker-image: "pytorch/pytorch:2.8.0-cuda12.9-cudnn9-devel" + runner: "mt-l-x86aavx2-11-41-l4" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.9.2-cudnn-devel-ubuntu24.04" timeout: 60 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then @@ -72,11 +72,11 @@ jobs: matrix: python_version: ["3.12"] cuda_arch_version: ["12.9"] - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g6.4xlarge.experimental.nvidia.gpu" - docker-image: "pytorch/pytorch:2.8.0-cuda12.9-cudnn9-devel" + runner: "mt-l-x86aavx2-11-41-l4" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.9.2-cudnn-devel-ubuntu24.04" timeout: 60 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then diff --git a/.github/workflows/test-linux-mujoco.yml b/.github/workflows/test-linux-mujoco.yml index 55df6a07cf7..e7e78504f2b 100644 --- a/.github/workflows/test-linux-mujoco.yml +++ b/.github/workflows/test-linux-mujoco.yml @@ -37,13 +37,13 @@ jobs: matrix: python_version: ["3.11"] cuda_arch_version: ["12.8"] - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "linux.g5.4xlarge.nvidia.gpu" + runner: "mt-l-x86aavx2-11-41-a10g" gpu-arch-type: cuda gpu-arch-version: "12.8" - docker-image: "nvidia/cuda:12.8.0-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.8.1-cudnn-devel-ubuntu24.04" timeout: 90 script: | if [[ "${{ github.ref }}" =~ release/* ]]; then diff --git a/.github/workflows/test-linux-sota.yml b/.github/workflows/test-linux-sota.yml index 64a2c512b0a..ae051475723 100644 --- a/.github/workflows/test-linux-sota.yml +++ b/.github/workflows/test-linux-sota.yml @@ -33,11 +33,11 @@ jobs: # filtering in .github/unittest/linux_sota/scripts/test_sota.py. shard: ["1", "2"] fail-fast: false - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.g5.4xlarge.nvidia.gpu + runner: mt-l-x86aavx2-11-41-a10g repository: pytorch/rl - docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04" gpu-arch-type: cuda gpu-arch-version: ${{ matrix.cuda_arch_version }} timeout: 90 diff --git a/.github/workflows/test-linux-tutorials.yml b/.github/workflows/test-linux-tutorials.yml index 41d5152cbe1..459e8455aa4 100644 --- a/.github/workflows/test-linux-tutorials.yml +++ b/.github/workflows/test-linux-tutorials.yml @@ -27,11 +27,11 @@ jobs: cuda_arch_version: ["13.0"] fail-fast: false # Run on all PRs and pushes to main/nightly/release branches - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.g5.4xlarge.nvidia.gpu + runner: mt-l-x86aavx2-11-41-a10g repository: pytorch/rl - docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04" gpu-arch-type: cuda gpu-arch-version: ${{ matrix.cuda_arch_version }} timeout: 120 diff --git a/.github/workflows/test-linux.yml b/.github/workflows/test-linux.yml index 5de7746380b..f6f6579e2dc 100644 --- a/.github/workflows/test-linux.yml +++ b/.github/workflows/test-linux.yml @@ -29,11 +29,11 @@ jobs: matrix: python_version: ["3.10", "3.14"] fail-fast: false - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.4xlarge + runner: mt-l-x86iavx512-16-128 repository: pytorch/rl - docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04" timeout: 90 script: | set -euo pipefail @@ -53,11 +53,11 @@ jobs: python_version: ["3.10", "3.13", "3.14"] shard: ["bulk", "collectors", "mp"] fail-fast: false - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.12xlarge + runner: mt-l-x86iavx512-48-384 repository: pytorch/rl - docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04" timeout: 120 upload-artifact: test-results-cpu-${{ matrix.python_version }}-shard-${{ matrix.shard }} script: | @@ -91,11 +91,11 @@ jobs: python_version: ["3.11", "3.12"] shard: ["bulk", "collectors", "mp"] fail-fast: false - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.12xlarge + runner: mt-l-x86iavx512-48-384 repository: pytorch/rl - docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04" timeout: 120 upload-artifact: test-results-cpu-${{ matrix.python_version }}-shard-${{ matrix.shard }} script: | @@ -129,11 +129,11 @@ jobs: # Shard 3: all other tests shard: ["1", "2", "3"] fail-fast: false - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.g5.4xlarge.nvidia.gpu + runner: mt-l-x86aavx2-11-41-a10g repository: pytorch/rl - docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04" gpu-arch-type: cuda gpu-arch-version: ${{ matrix.cuda_arch_version }} timeout: 120 @@ -173,11 +173,11 @@ jobs: python_version: ["3.12"] cuda_arch_version: ["13.0"] fail-fast: false - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.g5.4xlarge.nvidia.gpu + runner: mt-l-x86aavx2-11-41-a10g repository: pytorch/rl - docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04" gpu-arch-type: cuda gpu-arch-version: ${{ matrix.cuda_arch_version }} timeout: 120 @@ -222,11 +222,11 @@ jobs: # serial shards; see linux_olddeps/scripts_gym_0_13/run_test.sh. shard: ["transforms", "quarantine", "remainder"] fail-fast: false - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.g5.4xlarge.nvidia.gpu + runner: mt-l-x86aavx2-11-41-a10g repository: pytorch/rl - docker-image: "nvidia/cuda:11.8.0-cudnn8-devel-ubuntu22.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda11.8.0-cudnn8-devel-ubuntu22.04" gpu-arch-type: cuda gpu-arch-version: ${{ matrix.cuda_arch_version }} timeout: 120 @@ -265,11 +265,11 @@ jobs: python_version: ["3.12"] cuda_arch_version: ["13.0"] fail-fast: false - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.g5.4xlarge.nvidia.gpu + runner: mt-l-x86aavx2-11-41-a10g repository: pytorch/rl - docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04" gpu-arch-type: cuda gpu-arch-version: ${{ matrix.cuda_arch_version }} timeout: 120 @@ -306,11 +306,11 @@ jobs: python_version: ["3.12"] cuda_arch_version: ["13.0"] fail-fast: false - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.g5.4xlarge.nvidia.gpu + runner: mt-l-x86aavx2-11-41-a10g repository: pytorch/rl - docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04" gpu-arch-type: cuda gpu-arch-version: ${{ matrix.cuda_arch_version }} timeout: 60 @@ -343,11 +343,11 @@ jobs: # Test sharding: split tests into 3 parallel jobs for faster execution shard: ["1", "2", "3"] fail-fast: false - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.g5.4xlarge.nvidia.gpu + runner: mt-l-x86aavx2-11-41-a10g repository: pytorch/rl - docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04" gpu-arch-type: cuda gpu-arch-version: ${{ matrix.cuda_arch_version }} timeout: 120 @@ -389,11 +389,11 @@ jobs: python_version: ["3.12"] # "3.9", "3.10", "3.11" cuda_arch_version: ["13.0"] # "11.6", "11.7" fail-fast: false - uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main + uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: - runner: linux.g5.4xlarge.nvidia.gpu + runner: mt-l-x86aavx2-11-41-a10g repository: pytorch/rl - docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04" + docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04" gpu-arch-type: cuda gpu-arch-version: ${{ matrix.cuda_arch_version }} timeout: 120 From 1f80fc2311989f1df6ac2098aaeeaedc75f6e391 Mon Sep 17 00:00:00 2001 From: Huy Do Date: Thu, 8 Oct 2026 12:04:04 -0700 Subject: [PATCH 2/6] Use larger a10g runner for unittests-gym --- .github/workflows/test-linux-libs.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/test-linux-libs.yml b/.github/workflows/test-linux-libs.yml index 8d563e550b4..5f8380fe51b 100644 --- a/.github/workflows/test-linux-libs.yml +++ b/.github/workflows/test-linux-libs.yml @@ -360,7 +360,7 @@ jobs: uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "mt-l-x86aavx2-11-41-a10g" + runner: "mt-l-x86aavx2-29-113-a10g" # gpu-arch-type: "cuda" # gpu-arch-version: "11.6" docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" From 5ad118189860fc532b125e6d6f1864a12f15b527 Mon Sep 17 00:00:00 2001 From: Huy Do Date: Thu, 8 Oct 2026 14:09:06 -0700 Subject: [PATCH 3/6] Revert unittests-gym to the smaller a10g runner --- .github/workflows/test-linux-libs.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/test-linux-libs.yml b/.github/workflows/test-linux-libs.yml index 5f8380fe51b..8d563e550b4 100644 --- a/.github/workflows/test-linux-libs.yml +++ b/.github/workflows/test-linux-libs.yml @@ -360,7 +360,7 @@ jobs: uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main with: repository: pytorch/rl - runner: "mt-l-x86aavx2-29-113-a10g" + runner: "mt-l-x86aavx2-11-41-a10g" # gpu-arch-type: "cuda" # gpu-arch-version: "11.6" docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.4.1-cudnn-devel-ubuntu22.04" From fad60038e39a9693a0ba07e55f16238c85eaf05c Mon Sep 17 00:00:00 2001 From: Huy Do Date: Thu, 8 Oct 2026 14:19:25 -0700 Subject: [PATCH 4/6] [DEBUG] Instrument unittests-gym to diagnose OSDC pod failure --- .github/unittest/linux_libs/scripts_gym/run_all.sh | 6 +++++- .github/workflows/test-linux-libs.yml | 13 +++++++++++++ 2 files changed, 18 insertions(+), 1 deletion(-) diff --git a/.github/unittest/linux_libs/scripts_gym/run_all.sh b/.github/unittest/linux_libs/scripts_gym/run_all.sh index 4a2c5dd40d4..310a548caac 100755 --- a/.github/unittest/linux_libs/scripts_gym/run_all.sh +++ b/.github/unittest/linux_libs/scripts_gym/run_all.sh @@ -206,7 +206,11 @@ run_tests() { test_failed=1 fi - if ! python .github/unittest/helpers/coverage_run_parallel.py -m pytest test/libs --instafail -v --durations 200 -k "gym and not isaac" --mp_fork; then + # TODO: remove once the OSDC pod failure in test_gym_gymnasium_parallel is understood + python -m pytest test/libs/test_gym.py -v -s -o log_cli=true --log-cli-level=INFO -k test_gym_gymnasium_parallel --mp_fork || true + nvidia-smi || true + + if ! python .github/unittest/helpers/coverage_run_parallel.py -m pytest test/libs --instafail -v -o log_cli=true --durations 200 -k "gym and not isaac" --mp_fork; then echo "ERROR: test/libs failed for ${version_name}" test_failed=1 fi diff --git a/.github/workflows/test-linux-libs.yml b/.github/workflows/test-linux-libs.yml index 8d563e550b4..0d9988b3434 100644 --- a/.github/workflows/test-linux-libs.yml +++ b/.github/workflows/test-linux-libs.yml @@ -382,6 +382,19 @@ jobs: export TAR_OPTIONS="--no-same-owner" export BATCHED_PIPE_TIMEOUT=60 export TD_GET_DEFAULTS_TO_NONE=1 + export PYTHONUNBUFFERED=1 + + # TODO: remove once the OSDC pod failure in test_gym_gymnasium_parallel is understood + ulimit -a + cat /sys/fs/cgroup/memory.max /sys/fs/cgroup/pids.max || true + nvidia-smi || true + df -h /dev/shm + (set +x; while true; do + echo "[monitor] $(date -u +%T) gpu=$(nvidia-smi --query-gpu=memory.used,utilization.gpu --format=csv,noheader 2>&1)" \ + "mem=$(cat /sys/fs/cgroup/memory.current) pids=$(cat /sys/fs/cgroup/pids.current)" \ + "shm=$(df --output=used /dev/shm | tail -1) events=$(tr '\n' ' ' < /sys/fs/cgroup/memory.events)" + sleep 5 + done) & ./.github/unittest/linux_libs/scripts_gym/run_all.sh From 38e4c6b99bc06005d8014c8f19207e35978b13ff Mon Sep 17 00:00:00 2001 From: Huy Do Date: Thu, 8 Oct 2026 14:57:17 -0700 Subject: [PATCH 5/6] Use spawn for ParallelEnv in gym tests on CUDA machines --- .github/unittest/linux_libs/scripts_gym/run_all.sh | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/unittest/linux_libs/scripts_gym/run_all.sh b/.github/unittest/linux_libs/scripts_gym/run_all.sh index 310a548caac..5b9514c7a9a 100755 --- a/.github/unittest/linux_libs/scripts_gym/run_all.sh +++ b/.github/unittest/linux_libs/scripts_gym/run_all.sh @@ -210,7 +210,7 @@ run_tests() { python -m pytest test/libs/test_gym.py -v -s -o log_cli=true --log-cli-level=INFO -k test_gym_gymnasium_parallel --mp_fork || true nvidia-smi || true - if ! python .github/unittest/helpers/coverage_run_parallel.py -m pytest test/libs --instafail -v -o log_cli=true --durations 200 -k "gym and not isaac" --mp_fork; then + if ! python .github/unittest/helpers/coverage_run_parallel.py -m pytest test/libs --instafail -v -o log_cli=true --durations 200 -k "gym and not isaac" --mp_fork_if_no_cuda; then echo "ERROR: test/libs failed for ${version_name}" test_failed=1 fi From c1cfb1a1ddad1f5335a320f2d0f29350ec28fb01 Mon Sep 17 00:00:00 2001 From: Huy Do Date: Thu, 8 Oct 2026 14:57:26 -0700 Subject: [PATCH 6/6] Remove debug instrumentation from unittests-gym --- .github/unittest/linux_libs/scripts_gym/run_all.sh | 6 +----- .github/workflows/test-linux-libs.yml | 13 ------------- 2 files changed, 1 insertion(+), 18 deletions(-) diff --git a/.github/unittest/linux_libs/scripts_gym/run_all.sh b/.github/unittest/linux_libs/scripts_gym/run_all.sh index 5b9514c7a9a..822bc5a1947 100755 --- a/.github/unittest/linux_libs/scripts_gym/run_all.sh +++ b/.github/unittest/linux_libs/scripts_gym/run_all.sh @@ -206,11 +206,7 @@ run_tests() { test_failed=1 fi - # TODO: remove once the OSDC pod failure in test_gym_gymnasium_parallel is understood - python -m pytest test/libs/test_gym.py -v -s -o log_cli=true --log-cli-level=INFO -k test_gym_gymnasium_parallel --mp_fork || true - nvidia-smi || true - - if ! python .github/unittest/helpers/coverage_run_parallel.py -m pytest test/libs --instafail -v -o log_cli=true --durations 200 -k "gym and not isaac" --mp_fork_if_no_cuda; then + if ! python .github/unittest/helpers/coverage_run_parallel.py -m pytest test/libs --instafail -v --durations 200 -k "gym and not isaac" --mp_fork_if_no_cuda; then echo "ERROR: test/libs failed for ${version_name}" test_failed=1 fi diff --git a/.github/workflows/test-linux-libs.yml b/.github/workflows/test-linux-libs.yml index 0d9988b3434..8d563e550b4 100644 --- a/.github/workflows/test-linux-libs.yml +++ b/.github/workflows/test-linux-libs.yml @@ -382,19 +382,6 @@ jobs: export TAR_OPTIONS="--no-same-owner" export BATCHED_PIPE_TIMEOUT=60 export TD_GET_DEFAULTS_TO_NONE=1 - export PYTHONUNBUFFERED=1 - - # TODO: remove once the OSDC pod failure in test_gym_gymnasium_parallel is understood - ulimit -a - cat /sys/fs/cgroup/memory.max /sys/fs/cgroup/pids.max || true - nvidia-smi || true - df -h /dev/shm - (set +x; while true; do - echo "[monitor] $(date -u +%T) gpu=$(nvidia-smi --query-gpu=memory.used,utilization.gpu --format=csv,noheader 2>&1)" \ - "mem=$(cat /sys/fs/cgroup/memory.current) pids=$(cat /sys/fs/cgroup/pids.current)" \ - "shm=$(df --output=used /dev/shm | tail -1) events=$(tr '\n' ' ' < /sys/fs/cgroup/memory.events)" - sleep 5 - done) & ./.github/unittest/linux_libs/scripts_gym/run_all.sh