diff --git a/.github/actionlint.yaml b/.github/actionlint.yaml index 0c71349fbe..895b0b7246 100644 --- a/.github/actionlint.yaml +++ b/.github/actionlint.yaml @@ -1,5 +1,13 @@ self-hosted-runner: labels: - asv - - amdgpu - - nvidiagpu + - CPU + - AMDCPU + - INTELCPU + - GPU + - AMDGPU + - INTELGPU + - NVIDIAGPU + - MI50 + - MI210 + - V100 diff --git a/.github/workflows/docker-bases.yaml b/.github/workflows/docker-bases.yaml index c98b8c2ab3..bc5fa51b71 100644 --- a/.github/workflows/docker-bases.yaml +++ b/.github/workflows/docker-bases.yaml @@ -196,7 +196,7 @@ jobs: platform: linux/amd64 runner: - self-hosted - - nvidiagpu + - NVIDIAGPU - arch: arm64 platform: linux/arm64 runner: ubuntu-24.04-arm @@ -307,7 +307,7 @@ jobs: deploy-amd-bases: if: ${{ github.event_name != 'workflow_dispatch' || inputs.amd }} name: "amd-base" - runs-on: ["self-hosted", "amdgpu"] + runs-on: ["self-hosted", "AMDGPU"] env: DOCKER_BUILDKIT: "1" diff --git a/.github/workflows/docker-devito.yaml b/.github/workflows/docker-devito.yaml index 2f9fd42515..87ab0e212c 100644 --- a/.github/workflows/docker-devito.yaml +++ b/.github/workflows/docker-devito.yaml @@ -30,9 +30,9 @@ jobs: tag_suffix: '-amd64' # Respect CUDA_VISIBLE_DEVICES set by the runner and hard-limit docker to that device. # (--gpus maps only the selected device from CUDA_VISIBLE_DEVICES) - flag: --init --gpus "device=${CUDA_VISIBLE_DEVICES:-all}" + flag: --init --cpuset-cpus="${AVAILABLE_CPUSET}" --memory="${AVAILABLE_MEM}" --memory-swap="${AVAILABLE_MEM}" --gpus "device=${CUDA_VISIBLE_DEVICES}" test: 'tests/test_gpu_openacc.py tests/test_gpu_common.py' - runner: ["self-hosted", "nvidiagpu"] + runner: ["self-hosted", "NVIDIAGPU"] - base: 'bases:nvidia-nvc' tag: 'nvidia-nvc' @@ -52,9 +52,9 @@ jobs: tag_suffix: '-amd64' # Respect CUDA_VISIBLE_DEVICES set by the runner and hard-limit docker to that device. # (--gpus maps only the selected device from CUDA_VISIBLE_DEVICES) - flag: --init --gpus "device=${CUDA_VISIBLE_DEVICES:-all}" + flag: --init --cpuset-cpus="${AVAILABLE_CPUSET}" --memory="${AVAILABLE_MEM}" --memory-swap="${AVAILABLE_MEM}" --gpus "device=${CUDA_VISIBLE_DEVICES}" test: 'tests/test_gpu_openacc.py tests/test_gpu_common.py' - runner: ["self-hosted", "nvidiagpu"] + runner: ["self-hosted", "NVIDIAGPU"] - base: 'bases:nvidia-nvc12' tag: 'nvidia-nvc12' @@ -73,9 +73,9 @@ jobs: platform: linux/amd64 run_tests: true tag_suffix: '' - flag: '--init --network=host --device=/dev/kfd --device=/dev/dri --ipc=host --group-add video --group-add $(getent group render | cut -d: -f3) --cap-add=SYS_PTRACE --security-opt seccomp=unconfined' + flag: '--init --cpuset-cpus="${AVAILABLE_CPUSET}" --memory="${AVAILABLE_MEM}" --memory-swap="${AVAILABLE_MEM}" --network=host --device=/dev/kfd --device=/dev/dri/${HIP_DEVICE_NAME} --ipc=host --group-add video --group-add $(getent group render | cut -d: -f3) --cap-add=SYS_PTRACE --security-opt seccomp=unconfined' test: 'tests/test_gpu_openmp.py' - runner: ["self-hosted", "amdgpu"] + runner: ["self-hosted", "AMDGPU"] - base: 'bases:cpu-gcc' tag: "gcc" @@ -184,7 +184,7 @@ jobs: run: | docker run ${{ matrix.flag }} --rm -t --name "${CONTAINER_NAME}" \ devitocodes/devito:${{ matrix.tag }}-dev \ - pytest -v ${{ matrix.test }} + pytest -n auto --dist loadscope -v ${{ matrix.test }} deploy-devito-manifest: needs: deploy-devito @@ -276,13 +276,13 @@ jobs: matrix: include: - tag: 'nvidia-nvc' - flag: --init --gpus "device=${CUDA_VISIBLE_DEVICES:-all}" + flag: --init --cpuset-cpus="${AVAILABLE_CPUSET}" --memory="${AVAILABLE_MEM}" --memory-swap="${AVAILABLE_MEM}" --gpus "device=${CUDA_VISIBLE_DEVICES:-all}" test: 'tests/test_gpu_openacc.py tests/test_gpu_common.py' - runner: ["self-hosted", "nvidiagpu"] + runner: ["self-hosted", "NVIDIAGPU"] - tag: 'nvidia-nvc12' - flag: --init --gpus "device=${CUDA_VISIBLE_DEVICES:-all}" + flag: --init --cpuset-cpus="${AVAILABLE_CPUSET}" --memory="${AVAILABLE_MEM}" --memory-swap="${AVAILABLE_MEM}" --gpus "device=${CUDA_VISIBLE_DEVICES:-all}" test: 'tests/test_gpu_openacc.py tests/test_gpu_common.py' - runner: ["self-hosted", "nvidiagpu"] + runner: ["self-hosted", "NVIDIAGPU"] - tag: 'gcc' flag: '--init -t' test: 'tests/test_operator.py' @@ -307,4 +307,4 @@ jobs: run: | docker run ${{ matrix.flag }} --rm -t --name "${CONTAINER_NAME}" \ devitocodes/devito:${{ matrix.tag }}-dev \ - pytest -v ${{ matrix.test }} + pytest -n auto --dist loadscope -v ${{ matrix.test }} diff --git a/.github/workflows/pytest-gpu.yaml b/.github/workflows/pytest-gpu.yaml index 4259b94403..3ade43ecad 100644 --- a/.github/workflows/pytest-gpu.yaml +++ b/.github/workflows/pytest-gpu.yaml @@ -44,22 +44,29 @@ jobs: - name: pytest-gpu-acc-nvidia test_files: "tests/test_adjoint.py tests/test_gpu_common.py tests/test_gpu_openacc.py tests/test_operator.py::TestEstimateMemory" base: "devitocodes/bases:nvidia-nvc12" - runner_label: nvidiagpu + runner_label: NVIDIAGPU test_drive_cmd: "nvidia-smi" # Respect CUDA_VISIBLE_DEVICES and also hard-limit Docker to that device. # NOTE: CUDA_VISIBLE_DEVICES must be set by the runner (systemd drop-in etc.). - dockerflags: --gpus "device=${CUDA_VISIBLE_DEVICES:-all}" + dockerflags: >- + --cpuset-cpus="${AVAILABLE_CPUSET}" + --memory="${AVAILABLE_MEM}" + --memory-swap="${AVAILABLE_MEM}" + --gpus "device=${CUDA_VISIBLE_DEVICES}" # -------------------- AMD job ----------------------- - name: pytest-gpu-omp-amd test_files: "tests/test_adjoint.py tests/test_gpu_common.py tests/test_gpu_openmp.py tests/test_operator.py::TestEstimateMemory" - runner_label: amdgpu + runner_label: AMDGPU base: "devitocodes/bases:amd" test_drive_cmd: "rocm-smi" # Passes through required /dev nodes etc. dockerflags: >- + --cpuset-cpus="${AVAILABLE_CPUSET}" + --memory="${AVAILABLE_MEM}" + --memory-swap="${AVAILABLE_MEM}" --network=host - --device=/dev/kfd --device=/dev/dri + --device=/dev/kfd --device=/dev/dri/${HIP_DEVICE_NAME} --ipc=host --group-add video --group-add "$(getent group render | cut -d: -f3)" --cap-add=SYS_PTRACE --security-opt seccomp=unconfined @@ -76,6 +83,10 @@ jobs: tag: ${{ matrix.name }} base: ${{ matrix.base }} + - name: Runner environment + run: | + printenv + - name: Probe GPU uses: ./.github/actions/docker-run with: @@ -95,8 +106,10 @@ jobs: CODECOV_TOKEN=${{ secrets.CODECOV_TOKEN }} DEVITO_LOGGING=DEBUG PYTHONFAULTHANDLER=1 + PYTEST_XDIST_AUTO_NUM_WORKERS=${PYTEST_XDIST_AUTO_NUM_WORKERS} command: | pytest \ + -n auto --dist loadscope \ -vvv \ --capture=no \ --showlocals \ @@ -115,8 +128,10 @@ jobs: uid: ${{ steps.build.outputs.unique }} tag: ${{ matrix.name }} args: ${{ matrix.dockerflags }} + env: PYTEST_XDIST_AUTO_NUM_WORKERS=${PYTEST_XDIST_AUTO_NUM_WORKERS} command: | pytest \ + -n auto --dist loadscope \ -v \ ${{ matrix.test_examples }} diff --git a/requirements-testing.txt b/requirements-testing.txt index f2615758e3..58fc63050a 100644 --- a/requirements-testing.txt +++ b/requirements-testing.txt @@ -9,3 +9,4 @@ click<9.0 cloudpickle<3.1.3 ipympl<0.10.1 ipykernel<7.0.0 +pytest-xdist<3.8.1