Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
18 changes: 16 additions & 2 deletions .github/unittest/examples/scripts/run_all.sh
Original file line number Diff line number Diff line change
Expand Up @@ -12,12 +12,26 @@ apt-get install -y --no-install-recommends \
cmake curl ffmpeg g++ gcc git libegl1 libgl1 libgles2 libglfw3 libglvnd0 \
libglx-mesa0 libglew-dev libosmesa6 libosmesa6-dev python3-dev tzdata

# torchrl/_comm/distributed.py resolves socket.gethostbyname(gethostname()) to
# pick the dist.TCPStore address. Nothing puts an OSDC job container's hostname
# in its /etc/hosts, so that raises gaierror; a plain `docker run` added the
# entry, which is why this only appears off EC2. Every participant lives in
# this container, so loopback is the right answer.
echo "127.0.0.1 $(hostname)" >> /etc/hosts
getent hosts "$(hostname)"

this_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")" >/dev/null 2>&1 && pwd)"
root_dir="$(cd "${this_dir}/../../../.." >/dev/null 2>&1 && pwd)"
env_dir="${root_dir}/venv"

cp "${root_dir}/.github/unittest/tutorials/scripts/10_nvidia.json" \
/usr/share/glvnd/egl_vendor.d/10_nvidia.json
# from cudagl docker image. /usr/share/glvnd/egl_vendor.d is a read-only
# mount under the NVIDIA container runtime, so install into
# /etc/glvnd/egl_vendor.d, the other default libglvnd search path, and
# skip entirely when the runtime already supplied the file.
if [ ! -e /usr/share/glvnd/egl_vendor.d/10_nvidia.json ]; then
install -Dm644 "${root_dir}/.github/unittest/tutorials/scripts/10_nvidia.json" \
/etc/glvnd/egl_vendor.d/10_nvidia.json
fi
git config --global --add safe.directory '*'
cd "${root_dir}"

Expand Down
71 changes: 67 additions & 4 deletions .github/unittest/linux/scripts/run_all.sh
Original file line number Diff line number Diff line change
Expand Up @@ -39,10 +39,25 @@ if [[ $OSTYPE != 'darwin'* ]]; then
fi
fi

# torchrl/_comm/distributed.py resolves socket.gethostbyname(gethostname()) to
# pick the dist.TCPStore address. Nothing puts an OSDC job container's hostname
# in its /etc/hosts, so that raises gaierror; a plain `docker run` added the
# entry, which is why this only appears off EC2. Every participant lives in
# this container, so loopback is the right answer.
if [[ $OSTYPE != 'darwin'* ]]; then
echo "127.0.0.1 $(hostname)" >> /etc/hosts
getent hosts "$(hostname)"
fi

this_dir="$( cd "$( dirname "${BASH_SOURCE[0]}" )" >/dev/null 2>&1 && pwd )"
if [[ $OSTYPE != 'darwin'* ]]; then
# from cudagl docker image
cp $this_dir/10_nvidia.json /usr/share/glvnd/egl_vendor.d/10_nvidia.json
# from cudagl docker image. /usr/share/glvnd/egl_vendor.d is a read-only
# mount under the NVIDIA container runtime, so install into
# /etc/glvnd/egl_vendor.d, the other default libglvnd search path, and
# skip entirely when the runtime already supplied the file.
if [ ! -e /usr/share/glvnd/egl_vendor.d/10_nvidia.json ]; then
install -Dm644 $this_dir/10_nvidia.json /etc/glvnd/egl_vendor.d/10_nvidia.json
fi
fi


Expand Down Expand Up @@ -397,6 +412,23 @@ run_non_distributed_tests() {
test/test_inference_server.py
test/test_loggers.py
)
# Individual tests that must not run under xdist. The quarantine above is
# path-based, and moving all of test_dreamer_v3.py would cost ~1300s of the
# shard's 2950s, so deselect just the offender.
#
# test_dreamer_v3_checkpoint_resume_processes spawns a CUDA subprocess. With
# 16 xdist workers on one A10G it loses the race for a context and dies in
# cuDevicePrimaryCtxRetain with CUDA_ERROR_OUT_OF_MEMORY, before allocating
# anything.
local quarantine_test_ids=(
"test/objectives/test_dreamer_v3.py::test_dreamer_v3_checkpoint_resume_processes"
)
local quarantine_deselects=""
local quarantine_id
for quarantine_id in "${quarantine_test_ids[@]}"; do
quarantine_deselects+="--deselect ${quarantine_id} "
done

local quarantine_test_paths=("${collector_test_paths[@]}" "${mp_test_paths[@]}")
local collector_tests="${collector_test_paths[*]}"
local mp_tests="${mp_test_paths[*]}"
Expand Down Expand Up @@ -439,7 +471,13 @@ run_non_distributed_tests() {
# configuration the timing numbers were collected with.
xdist_workers=24
else
xdist_workers=auto
# GPU: bound by device memory, not cores. -n auto gives 16 workers here,
# and 16 concurrent CUDA contexts on the single A10G leave no room for a
# test that spawns its own CUDA subprocess -- they die in
# cuDevicePrimaryCtxRetain with CUDA_ERROR_OUT_OF_MEMORY, and ordinary
# Triton tests start failing to allocate too. Which tests lose the race
# varies run to run, so cap the concurrency rather than chase them.
xdist_workers=4
fi
fi
local xdist_args=""
Expand Down Expand Up @@ -469,7 +507,7 @@ run_non_distributed_tests() {
;;
2)
echo "Running shard 2: process-spawning tests (${quarantine_tests})"
python .github/unittest/helpers/coverage_run_parallel.py -m pytest ${quarantine_tests} \
python .github/unittest/helpers/coverage_run_parallel.py -m pytest ${quarantine_tests} "${quarantine_test_ids[@]}" \
"${GPU_MARKER_FILTER[@]}" \
${json_report_args} \
${common_args} ${serial_timeout}
Expand All @@ -480,6 +518,7 @@ run_non_distributed_tests() {
${common_ignores} \
--ignore test/transforms \
${quarantine_ignores} \
${quarantine_deselects} \
${xdist_args} \
"${GPU_MARKER_FILTER[@]}" \
${json_report_args} \
Expand All @@ -493,6 +532,7 @@ run_non_distributed_tests() {
python .github/unittest/helpers/coverage_run_parallel.py -m pytest test \
${common_ignores} \
${quarantine_ignores} \
${quarantine_deselects} \
${xdist_args} \
"${GPU_MARKER_FILTER[@]}" \
${json_report_args} \
Expand Down Expand Up @@ -594,5 +634,28 @@ python .github/unittest/helpers/upload_test_results.py || echo "Warning: Failed

bash ${this_dir}/post_process.sh

# ==================================================================================== #
# ================================ Reap strays ======================================= #

# On OSDC the step runs under run_with_env_secrets.py, which drains our stdout
# until EOF. EOF only arrives once every process holding the write end has
# closed it, so a single xdist worker or Ray actor that outlives pytest keeps
# the job alive until its 120-minute timeout, long after this script exits 0.
# Under linux_job_v2 the surrounding `docker run` reaped these for us.
echo "::group::Processes still alive before exit"
ps -eo pid,ppid,etimes,rss,args --sort=-rss | head -40 || true
echo "::endgroup::"

# Anything still running that is not this shell or ps itself.
strays="$(pgrep -f 'pytest|ray::|Xvfb' 2>/dev/null | grep -v "^$$\$" || true)"
if [ -n "${strays}" ]; then
echo "Reaping strays holding the step open: ${strays}"
# shellcheck disable=SC2086
kill -TERM ${strays} 2>/dev/null || true
sleep 5
# shellcheck disable=SC2086
kill -KILL ${strays} 2>/dev/null || true
fi

# Exit with failure if any tests failed
exit $EXIT_STATUS
13 changes: 10 additions & 3 deletions .github/unittest/linux_libs/scripts_habitat/run_all.sh
Original file line number Diff line number Diff line change
Expand Up @@ -8,7 +8,9 @@ apt-get update
apt-get install -y vim git wget cmake ninja-build

# OpenGL/EGL dependencies for headless rendering
apt-get install -y libglfw3 libglfw3-dev libgl1-mesa-glx libosmesa6 libosmesa6-dev libglew-dev
# libgl1-mesa-glx is gone on noble; libgl1 + libglx-mesa0 is what it became,
# and both resolve on jammy too.
apt-get install -y libglfw3 libglfw3-dev libgl1 libglx-mesa0 libosmesa6 libosmesa6-dev libglew-dev
apt-get install -y libxinerama-dev libxcursor-dev libxi-dev libxrandr-dev libxxf86vm-dev
apt-get install -y libglvnd0 libgl1 libglx0 libegl1 libgles2
apt-get install -y libegl1-mesa-dev libgles2-mesa-dev
Expand All @@ -21,8 +23,13 @@ apt-get install -y pkg-config
#apt-get upgrade -y libstdc++6
#apt-get install -y libgcc
this_dir="$( cd "$( dirname "${BASH_SOURCE[0]}" )" >/dev/null 2>&1 && pwd )"
# from cudagl docker image
cp $this_dir/10_nvidia.json /usr/share/glvnd/egl_vendor.d/10_nvidia.json
# from cudagl docker image. /usr/share/glvnd/egl_vendor.d is a read-only
# mount under the NVIDIA container runtime, so install into
# /etc/glvnd/egl_vendor.d, the other default libglvnd search path, and
# skip entirely when the runtime already supplied the file.
if [ ! -e /usr/share/glvnd/egl_vendor.d/10_nvidia.json ]; then
install -Dm644 $this_dir/10_nvidia.json /etc/glvnd/egl_vendor.d/10_nvidia.json
fi

bash ${this_dir}/setup_env.sh
bash ${this_dir}/install.sh
Expand Down
5 changes: 4 additions & 1 deletion .github/unittest/linux_libs/scripts_mujoco/run_all.sh
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,10 @@
set -euxo pipefail

apt update
apt install -y libglfw3 libglfw3-dev libglew-dev libgl1-mesa-glx libgl1-mesa-dev mesa-common-dev libegl1-mesa-dev freeglut3 freeglut3-dev
# libgl1-mesa-glx and freeglut3 are gone on noble; libgl1 + libglx-mesa0 is
# what the former became, and freeglut3-dev pulls the runtime. All of these
# resolve on jammy too, so the list works on either image.
apt install -y libglfw3 libglfw3-dev libglew-dev libgl1 libglx-mesa0 libgl1-mesa-dev mesa-common-dev libegl1-mesa-dev freeglut3-dev

this_dir="$( cd "$( dirname "${BASH_SOURCE[0]}" )" >/dev/null 2>&1 && pwd )"
bash ${this_dir}/setup_env.sh
Expand Down
3 changes: 2 additions & 1 deletion .github/unittest/linux_libs/scripts_mujoco/setup_env.sh
Original file line number Diff line number Diff line change
Expand Up @@ -17,7 +17,8 @@ apt-get install -y wget \
curl \
patchelf \
libosmesa6-dev \
libgl1-mesa-glx \
libgl1 \
libglx-mesa0 \
libglfw3 \
libglew-dev \
libglvnd0 \
Expand Down
9 changes: 7 additions & 2 deletions .github/unittest/linux_optdeps/scripts/run_all.sh
Original file line number Diff line number Diff line change
Expand Up @@ -39,8 +39,13 @@ fi

this_dir="$( cd "$( dirname "${BASH_SOURCE[0]}" )" >/dev/null 2>&1 && pwd )"
if [[ $OSTYPE != 'darwin'* ]]; then
# from cudagl docker image
cp $this_dir/10_nvidia.json /usr/share/glvnd/egl_vendor.d/10_nvidia.json
# from cudagl docker image. /usr/share/glvnd/egl_vendor.d is a read-only
# mount under the NVIDIA container runtime, so install into
# /etc/glvnd/egl_vendor.d, the other default libglvnd search path, and
# skip entirely when the runtime already supplied the file.
if [ ! -e /usr/share/glvnd/egl_vendor.d/10_nvidia.json ]; then
install -Dm644 $this_dir/10_nvidia.json /etc/glvnd/egl_vendor.d/10_nvidia.json
fi
fi


Expand Down
9 changes: 7 additions & 2 deletions .github/unittest/linux_sota/scripts/run_all.sh
Original file line number Diff line number Diff line change
Expand Up @@ -23,8 +23,13 @@ apt-get install -y libglvnd0 libgl1 libglx0 libglx-mesa0 libegl1 libgles2
apt-get install -y g++ gcc patchelf

this_dir="$( cd "$( dirname "${BASH_SOURCE[0]}" )" >/dev/null 2>&1 && pwd )"
# from cudagl docker image
cp $this_dir/10_nvidia.json /usr/share/glvnd/egl_vendor.d/10_nvidia.json
# from cudagl docker image. /usr/share/glvnd/egl_vendor.d is a read-only
# mount under the NVIDIA container runtime, so install into
# /etc/glvnd/egl_vendor.d, the other default libglvnd search path, and
# skip entirely when the runtime already supplied the file.
if [ ! -e /usr/share/glvnd/egl_vendor.d/10_nvidia.json ]; then
install -Dm644 $this_dir/10_nvidia.json /etc/glvnd/egl_vendor.d/10_nvidia.json
fi


# ==================================================================================== #
Expand Down
9 changes: 7 additions & 2 deletions .github/unittest/tutorials/scripts/run_all.sh
Original file line number Diff line number Diff line change
Expand Up @@ -25,8 +25,13 @@ apt-get install -y ffmpeg libavcodec-dev libavformat-dev libavutil-dev libswscal
apt-get install -y libavdevice-dev libavfilter-dev libswresample-dev pkg-config

this_dir="$( cd "$( dirname "${BASH_SOURCE[0]}" )" >/dev/null 2>&1 && pwd )"
# from cudagl docker image
cp $this_dir/10_nvidia.json /usr/share/glvnd/egl_vendor.d/10_nvidia.json
# from cudagl docker image. /usr/share/glvnd/egl_vendor.d is a read-only
# mount under the NVIDIA container runtime, so install into
# /etc/glvnd/egl_vendor.d, the other default libglvnd search path, and
# skip entirely when the runtime already supplied the file.
if [ ! -e /usr/share/glvnd/egl_vendor.d/10_nvidia.json ]; then
install -Dm644 $this_dir/10_nvidia.json /etc/glvnd/egl_vendor.d/10_nvidia.json
fi


# ==================================================================================== #
Expand Down
2 changes: 1 addition & 1 deletion .github/workflows/benchmarks.yml
Original file line number Diff line number Diff line change
Expand Up @@ -66,7 +66,7 @@ jobs:
if: github.event_name != 'pull_request'
needs: validate-report
name: ${{ matrix.device }} Pytest benchmark
runs-on: linux.g5.4xlarge.nvidia.gpu
runs-on: mt-l-x86aavx2-11-41-a10g
timeout-minutes: 120
strategy:
fail-fast: false
Expand Down
6 changes: 3 additions & 3 deletions .github/workflows/benchmarks_pr.yml
Original file line number Diff line number Diff line change
Expand Up @@ -71,7 +71,7 @@ jobs:
prepare-environment:
name: Prepare pinned benchmark environment
if: contains(github.event.pull_request.labels.*.name, 'benchmarks/upload')
runs-on: linux.g5.4xlarge.nvidia.gpu
runs-on: mt-l-x86aavx2-11-41-a10g
container:
image: nvidia/cuda:12.6.3-cudnn-devel-ubuntu22.04@sha256:b3e7fba84d169f46939f00c25be7d016f712a8d651f4756d6a55e693d84d94f2
options: --gpus all --shm-size=8g
Expand Down Expand Up @@ -153,7 +153,7 @@ jobs:
name: ${{ matrix.device }} ${{ matrix.revision }} benchmark
if: contains(github.event.pull_request.labels.*.name, 'benchmarks/upload')
needs: [prepare-definitions, prepare-environment]
runs-on: linux.g5.4xlarge.nvidia.gpu
runs-on: mt-l-x86aavx2-11-41-a10g
strategy:
fail-fast: false
max-parallel: 4
Expand Down Expand Up @@ -315,7 +315,7 @@ jobs:
"pr_number": int(os.environ["PR_NUMBER"]),
"base_sha": os.environ["BASE_SHA"],
"head_sha": os.environ["HEAD_SHA"],
"runner": "linux.g5.4xlarge.nvidia.gpu",
"runner": "mt-l-x86aavx2-11-41-a10g",
"image": os.environ["IMAGE"],
"python_version": os.environ["PYTHON_VERSION"],
"system_environment_sha256": os.environ["SYSTEM_ENVIRONMENT_SHA"],
Expand Down
6 changes: 4 additions & 2 deletions .github/workflows/lint.yml
Original file line number Diff line number Diff line change
Expand Up @@ -22,8 +22,9 @@ permissions:

jobs:
python-source-and-configs:
uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main
uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main
with:
runner: mt-l-x86iavx512-8-64
repository: pytorch/rl
script: |
set -euo pipefail
Expand All @@ -50,8 +51,9 @@ jobs:
echo '::endgroup::'

c-source:
uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main
uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main
with:
runner: mt-l-x86iavx512-8-64
repository: pytorch/rl
script: |
set -euo pipefail
Expand Down
6 changes: 3 additions & 3 deletions .github/workflows/test-linux-examples.yml
Original file line number Diff line number Diff line change
Expand Up @@ -26,11 +26,11 @@ jobs:
cuda_arch_version: ["13.0"]
shard: ["1", "2"]
fail-fast: false
uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main
uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main
with:
runner: linux.g5.4xlarge.nvidia.gpu
runner: mt-l-x86aavx2-11-41-a10g
repository: pytorch/rl
docker-image: "nvidia/cuda:13.0.2-cudnn-devel-ubuntu24.04"
docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda13.0.3-cudnn-devel-ubuntu24.04"
gpu-arch-type: cuda
gpu-arch-version: ${{ matrix.cuda_arch_version }}
timeout: 120
Expand Down
6 changes: 3 additions & 3 deletions .github/workflows/test-linux-habitat.yml
Original file line number Diff line number Diff line change
Expand Up @@ -28,11 +28,11 @@ jobs:
python_version: ["3.10"]
cuda_arch_version: ["12.8"]
fail-fast: false
uses: pytorch/test-infra/.github/workflows/linux_job_v2.yml@main
uses: pytorch/test-infra/.github/workflows/linux_job_v3.yml@main
with:
runner: linux.g5.12xlarge.nvidia.gpu
runner: mt-l-x86aavx2-45-167-a10g-4
repository: pytorch/rl
docker-image: "nvidia/cuda:12.8.1-cudnn-devel-ubuntu22.04"
docker-image: "ghcr.io/pytorch/test-infra/osdc-cuda:cuda12.8.1-cudnn-devel-ubuntu24.04"
gpu-arch-type: cuda
gpu-arch-version: ${{ matrix.cuda_arch_version }}
timeout: 90
Expand Down
Loading
Loading