Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
40 changes: 40 additions & 0 deletions helm/slurm-cluster/slurm_scripts/check_nvidia_libraries.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,40 @@
#!/bin/bash

set -euxo pipefail

fail_if_lib_dne () {
checklib=$1
if python -c "import ctypes; ctypes.CDLL('$checklib')"; then
echo "...found $checklib, continuing"
else
echo "ERROR: missing $checklib"
exit 1
fi
}

# first check if I can load libnvidia-ml.so.1, because nothing works without it
# (also not a typo, looks specifically for .1)

fail_if_lib_dne "libnvidia-ml.so.1"

# get list of base Nvidia libraries to check from nvidia-container-cli
full_library_list=$(nvidia-container-cli list -l | grep \\.so)
for x in $full_library_list; do
fail_if_lib_dne "$x"
done

# check against a list of commonly needed CUDA libraries that are common in a
# standard Pytorch training session

cudalibs="libcuda libcudart libcublas libcublasLt libcufft libcurand libcusolver libcusparse libcudnn libnccl libucp libucs libuct libgdrapi libmpi libnuma"
libs_w_version="libibverbs.so.1 librdmacm.so.1 libgomp.so.1"

for x in $cudalibs; do
fail_if_lib_dne "$x.so"
done
for x in $libs_w_version; do
fail_if_lib_dne "$x"
done

echo "OK: all libraries exist"
exit 0
17 changes: 17 additions & 0 deletions helm/slurm-cluster/slurm_scripts/check_nvidia_libraries.sh.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,17 @@
{
"name": "check_nvidia_libraries",
"command": "./check_nvidia_libraries.sh",
"platforms": ["8xGPU", "1xGPU", "4xGPU"],
"skip_for_cpu_jobs": false,
"skip_for_partial_gpu_jobs": false,
"skip_for_reservation_prefixes": [],
"contexts": ["prolog", "hc_program"],
"node_states": ["any"],
"on_fail": "drain",
"on_ok": "undrain",
"reason_base": "[node_problem] $name",

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

All nodes drained with [node_problem] prefix are being automatically replaced by Soperator automation.
But these drains show issues with hardware, not software.

This check doesn't show hardware issues, so automatic node replacement isn't needed. I suggest to change the prefix to something else, e.g. [software_problem]

"reason_append_details": true,
"run_in_jail": true,
"log": "slurm_scripts/$worker.$name.$context.out",
"need_env": []
}
4 changes: 4 additions & 0 deletions helm/slurm-cluster/values.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -744,6 +744,10 @@ slurmScripts:
enabled: true
customContent: null
customConfig: null
check_nvidia_libraries.sh:
enabled: true
customContent: null
customConfig: null
chmod_enroot_layers.sh:
enabled: true
customContent: null
Expand Down
Loading