Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/unittest/linux_libs/scripts_gym/run_all.sh
Original file line number Diff line number Diff line change
Expand Up @@ -206,7 +206,7 @@ run_tests() {
test_failed=1
fi

if ! python .github/unittest/helpers/coverage_run_parallel.py -m pytest test/libs --instafail -v --durations 200 -k "gym and not isaac" --mp_fork; then
if ! python .github/unittest/helpers/coverage_run_parallel.py -m pytest test/libs --instafail -v --durations 200 -k "gym and not isaac" --mp_fork_if_no_cuda; then
echo "ERROR: test/libs failed for ${version_name}"
test_failed=1
fi
Expand Down
2 changes: 1 addition & 1 deletion .github/workflows/benchmarks.yml
Original file line number Diff line number Diff line change
Expand Up @@ -66,7 +66,7 @@ jobs:
if: github.event_name != 'pull_request'
needs: validate-report
name: ${{ matrix.device }} Pytest benchmark
runs-on: linux.g5.4xlarge.nvidia.gpu
runs-on: mt-l-x86aavx2-11-41-a10g
timeout-minutes: 120
strategy:
fail-fast: false
Expand Down
6 changes: 3 additions & 3 deletions .github/workflows/benchmarks_pr.yml
Original file line number Diff line number Diff line change
Expand Up @@ -71,7 +71,7 @@ jobs:
prepare-environment:
name: Prepare pinned benchmark environment
if: contains(github.event.pull_request.labels.*.name, 'benchmarks/upload')
runs-on: linux.g5.4xlarge.nvidia.gpu
runs-on: mt-l-x86aavx2-11-41-a10g
container:
image: nvidia/cuda:12.6.3-cudnn-devel-ubuntu22.04@sha256:b3e7fba84d169f46939f00c25be7d016f712a8d651f4756d6a55e693d84d94f2
options: --gpus all --shm-size=8g
Expand Down Expand Up @@ -153,7 +153,7 @@ jobs:
name: ${{ matrix.device }} ${{ matrix.revision }} benchmark
if: contains(github.event.pull_request.labels.*.name, 'benchmarks/upload')
needs: [prepare-definitions, prepare-environment]
runs-on: linux.g5.4xlarge.nvidia.gpu
runs-on: mt-l-x86aavx2-11-41-a10g
strategy:
fail-fast: false
max-parallel: 4
Expand Down Expand Up @@ -315,7 +315,7 @@ jobs:
"pr_number": int(os.environ["PR_NUMBER"]),
"base_sha": os.environ["BASE_SHA"],
"head_sha": os.environ["HEAD_SHA"],
"runner": "linux.g5.4xlarge.nvidia.gpu",
"runner": "mt-l-x86aavx2-11-41-a10g",
"image": os.environ["IMAGE"],
"python_version": os.environ["PYTHON_VERSION"],
"system_environment_sha256": os.environ["SYSTEM_ENVIRONMENT_SHA"],
Expand Down
6 changes: 4 additions & 2 deletions .github/workflows/lint.yml
Original file line number Diff line number Diff line change
Expand Up @@ -22,8 +22,9 @@ permissions:

jobs:
python-source-and-configs:
uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main
uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main
with:
runner: mt-l-x86iavx512-8-64
repository: pytorch/rl
script: |
set -euo pipefail
Expand All @@ -50,8 +51,9 @@ jobs:
echo '::endgroup::'

c-source:
uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main
uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main
with:
runner: mt-l-x86iavx512-8-64
repository: pytorch/rl
script: |
set -euo pipefail
Expand Down
6 changes: 3 additions & 3 deletions .github/workflows/test-linux-examples.yml
Original file line number Diff line number Diff line change
Expand Up @@ -26,11 +26,11 @@ jobs:
cuda_arch_version: ["13.0"]
shard: ["1", "2"]
fail-fast: false
uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main
uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main
with:
runner: linux.g5.4xlarge.nvidia.gpu
runner: mt-l-x86aavx2-11-41-a10g
repository: pytorch/rl
docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04"
docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04"
gpu-arch-type: cuda
gpu-arch-version: ${{ matrix.cuda_arch_version }}
timeout: 120
Expand Down
6 changes: 3 additions & 3 deletions .github/workflows/test-linux-habitat.yml
Original file line number Diff line number Diff line change
Expand Up @@ -28,11 +28,11 @@ jobs:
python_version: ["3.10"]
cuda_arch_version: ["12.8"]
fail-fast: false
uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main
uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main
with:
runner: linux.g5.12xlarge.nvidia.gpu
runner: mt-l-x86aavx2-45-167-a10g-4
repository: pytorch/rl
docker-image: "nvidia/cuda:12.8.1-cudnn-devel-ubuntu22.04"
docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.8.1-cudnn-devel-ubuntu24.04"
gpu-arch-type: cuda
gpu-arch-version: ${{ matrix.cuda_arch_version }}
timeout: 90
Expand Down
260 changes: 151 additions & 109 deletions .github/workflows/test-linux-libs.yml

Large diffs are not rendered by default.

12 changes: 6 additions & 6 deletions .github/workflows/test-linux-llm.yml
Original file line number Diff line number Diff line change
Expand Up @@ -29,11 +29,11 @@ jobs:
matrix:
python_version: ["3.12"]
cuda_arch_version: ["12.9"]
uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main
uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main
with:
repository: pytorch/rl
runner: "linux.g6.4xlarge.experimental.nvidia.gpu"
docker-image: "pytorch/pytorch:2.8.0-cuda12.9-cudnn9-devel"
runner: "mt-l-x86aavx2-11-41-l4"
docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.9.2-cudnn-devel-ubuntu24.04"
timeout: 60
script: |
if [[ "${{ github.ref }}" =~ release/* ]]; then
Expand Down Expand Up @@ -72,11 +72,11 @@ jobs:
matrix:
python_version: ["3.12"]
cuda_arch_version: ["12.9"]
uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main
uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main
with:
repository: pytorch/rl
runner: "linux.g6.4xlarge.experimental.nvidia.gpu"
docker-image: "pytorch/pytorch:2.8.0-cuda12.9-cudnn9-devel"
runner: "mt-l-x86aavx2-11-41-l4"
docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.9.2-cudnn-devel-ubuntu24.04"
timeout: 60
script: |
if [[ "${{ github.ref }}" =~ release/* ]]; then
Expand Down
6 changes: 3 additions & 3 deletions .github/workflows/test-linux-mujoco.yml
Original file line number Diff line number Diff line change
Expand Up @@ -37,13 +37,13 @@ jobs:
matrix:
python_version: ["3.11"]
cuda_arch_version: ["12.8"]
uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main
uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main
with:
repository: pytorch/rl
runner: "linux.g5.4xlarge.nvidia.gpu"
runner: "mt-l-x86aavx2-11-41-a10g"
gpu-arch-type: cuda
gpu-arch-version: "12.8"
docker-image: "nvidia/cuda:12.8.0-devel-ubuntu22.04"
docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.8.1-cudnn-devel-ubuntu24.04"
timeout: 90
script: |
if [[ "${{ github.ref }}" =~ release/* ]]; then
Expand Down
6 changes: 3 additions & 3 deletions .github/workflows/test-linux-sota.yml
Original file line number Diff line number Diff line change
Expand Up @@ -33,11 +33,11 @@ jobs:
# filtering in .github/unittest/linux_sota/scripts/test_sota.py.
shard: ["1", "2"]
fail-fast: false
uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main
uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main
with:
runner: linux.g5.4xlarge.nvidia.gpu
runner: mt-l-x86aavx2-11-41-a10g
repository: pytorch/rl
docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04"
docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04"
gpu-arch-type: cuda
gpu-arch-version: ${{ matrix.cuda_arch_version }}
timeout: 90
Expand Down
6 changes: 3 additions & 3 deletions .github/workflows/test-linux-tutorials.yml
Original file line number Diff line number Diff line change
Expand Up @@ -27,11 +27,11 @@ jobs:
cuda_arch_version: ["13.0"]
fail-fast: false
# Run on all PRs and pushes to main/nightly/release branches
uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main
uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main
with:
runner: linux.g5.4xlarge.nvidia.gpu
runner: mt-l-x86aavx2-11-41-a10g
repository: pytorch/rl
docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04"
docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04"
gpu-arch-type: cuda
gpu-arch-version: ${{ matrix.cuda_arch_version }}
timeout: 120
Expand Down
60 changes: 30 additions & 30 deletions .github/workflows/test-linux.yml
Original file line number Diff line number Diff line change
Expand Up @@ -29,11 +29,11 @@ jobs:
matrix:
python_version: ["3.10", "3.14"]
fail-fast: false
uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main
uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main
with:
runner: linux.4xlarge
runner: mt-l-x86iavx512-16-128
repository: pytorch/rl
docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04"
docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04"
timeout: 90
script: |
set -euo pipefail
Expand All @@ -53,11 +53,11 @@ jobs:
python_version: ["3.10", "3.13", "3.14"]
shard: ["bulk", "collectors", "mp"]
fail-fast: false
uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main
uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main
with:
runner: linux.12xlarge
runner: mt-l-x86iavx512-48-384
repository: pytorch/rl
docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04"
docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04"
timeout: 120
upload-artifact: test-results-cpu-${{ matrix.python_version }}-shard-${{ matrix.shard }}
script: |
Expand Down Expand Up @@ -91,11 +91,11 @@ jobs:
python_version: ["3.11", "3.12"]
shard: ["bulk", "collectors", "mp"]
fail-fast: false
uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main
uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main
with:
runner: linux.12xlarge
runner: mt-l-x86iavx512-48-384
repository: pytorch/rl
docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04"
docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04"
timeout: 120
upload-artifact: test-results-cpu-${{ matrix.python_version }}-shard-${{ matrix.shard }}
script: |
Expand Down Expand Up @@ -129,11 +129,11 @@ jobs:
# Shard 3: all other tests
shard: ["1", "2", "3"]
fail-fast: false
uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main
uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main
with:
runner: linux.g5.4xlarge.nvidia.gpu
runner: mt-l-x86aavx2-11-41-a10g
repository: pytorch/rl
docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04"
docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04"
gpu-arch-type: cuda
gpu-arch-version: ${{ matrix.cuda_arch_version }}
timeout: 120
Expand Down Expand Up @@ -173,11 +173,11 @@ jobs:
python_version: ["3.12"]
cuda_arch_version: ["13.0"]
fail-fast: false
uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main
uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main
with:
runner: linux.g5.4xlarge.nvidia.gpu
runner: mt-l-x86aavx2-11-41-a10g
repository: pytorch/rl
docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04"
docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04"
gpu-arch-type: cuda
gpu-arch-version: ${{ matrix.cuda_arch_version }}
timeout: 120
Expand Down Expand Up @@ -222,11 +222,11 @@ jobs:
# serial shards; see linux_olddeps/scripts_gym_0_13/run_test.sh.
shard: ["transforms", "quarantine", "remainder"]
fail-fast: false
uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main
uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main
with:
runner: linux.g5.4xlarge.nvidia.gpu
runner: mt-l-x86aavx2-11-41-a10g
repository: pytorch/rl
docker-image: "nvidia/cuda:11.8.0-cudnn8-devel-ubuntu22.04"
docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda11.8.0-cudnn8-devel-ubuntu22.04"
gpu-arch-type: cuda
gpu-arch-version: ${{ matrix.cuda_arch_version }}
timeout: 120
Expand Down Expand Up @@ -265,11 +265,11 @@ jobs:
python_version: ["3.12"]
cuda_arch_version: ["13.0"]
fail-fast: false
uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main
uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main
with:
runner: linux.g5.4xlarge.nvidia.gpu
runner: mt-l-x86aavx2-11-41-a10g
repository: pytorch/rl
docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04"
docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04"
gpu-arch-type: cuda
gpu-arch-version: ${{ matrix.cuda_arch_version }}
timeout: 120
Expand Down Expand Up @@ -306,11 +306,11 @@ jobs:
python_version: ["3.12"]
cuda_arch_version: ["13.0"]
fail-fast: false
uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main
uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main
with:
runner: linux.g5.4xlarge.nvidia.gpu
runner: mt-l-x86aavx2-11-41-a10g
repository: pytorch/rl
docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04"
docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04"
gpu-arch-type: cuda
gpu-arch-version: ${{ matrix.cuda_arch_version }}
timeout: 60
Expand Down Expand Up @@ -343,11 +343,11 @@ jobs:
# Test sharding: split tests into 3 parallel jobs for faster execution
shard: ["1", "2", "3"]
fail-fast: false
uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main
uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main
with:
runner: linux.g5.4xlarge.nvidia.gpu
runner: mt-l-x86aavx2-11-41-a10g
repository: pytorch/rl
docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04"
docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04"
gpu-arch-type: cuda
gpu-arch-version: ${{ matrix.cuda_arch_version }}
timeout: 120
Expand Down Expand Up @@ -389,11 +389,11 @@ jobs:
python_version: ["3.12"] # "3.9", "3.10", "3.11"
cuda_arch_version: ["13.0"] # "11.6", "11.7"
fail-fast: false
uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main
uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main
with:
runner: linux.g5.4xlarge.nvidia.gpu
runner: mt-l-x86aavx2-11-41-a10g
repository: pytorch/rl
docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04"
docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04"
gpu-arch-type: cuda
gpu-arch-version: ${{ matrix.cuda_arch_version }}
timeout: 120
Expand Down
Loading