Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
22 changes: 21 additions & 1 deletion cmax/scripts/1-audit/cluster-audit-slurm.sh
Original file line number Diff line number Diff line change
Expand Up @@ -2912,9 +2912,29 @@ if systemctl is-active node-exporter prometheus-node-exporter &>/dev/null; then
# and unprivileged `ss -p` may hide process details. The executable comm name
# remains visible through procfs.
if pgrep -x grafana &>/dev/null || pgrep -x grafana-server &>/dev/null; then GRAFANA_DETECTED="true"; fi
# Slurm on Kubernetes runs the head node as a pod. The GPU Operator runs
# dcgm-exporter as a DaemonSet in other pods on the GPU nodes, so this pod has
# no listener on :9400 and no systemd unit. Query the exporter Service and
# count it only when it serves DCGM metrics. Cluster search domains in
# resolv.conf identify a pod: `su -` login shells drop KUBERNETES_SERVICE_HOST.
# CLUSTERMAX_DCGM_EXPORTER_URL sets the URL for other namespaces or exporters.
DCGM_EXPORTER_EVIDENCE="port 9400 or systemd active"
if [[ "$DCGM_EXPORTER_DETECTED" != "true" ]] && command -v curl &>/dev/null; then
DCGM_EXPORTER_URL="${CLUSTERMAX_DCGM_EXPORTER_URL:-}"
if [[ -z "$DCGM_EXPORTER_URL" ]] \
&& grep -qE '^search[[:space:]].*\.svc\.' "${CLUSTERMAX_RESOLV_CONF:-/etc/resolv.conf}" 2>/dev/null; then
DCGM_EXPORTER_URL="http://nvidia-dcgm-exporter.gpu-operator.svc:9400/metrics"
fi
if [[ -n "$DCGM_EXPORTER_URL" ]] \
&& DCGM_EXPORTER_METRICS=$(curl -fsS --max-time 5 "$DCGM_EXPORTER_URL" 2>/dev/null) \
&& grep -q '^DCGM_FI_' <<<"$DCGM_EXPORTER_METRICS"; then
DCGM_EXPORTER_DETECTED="true"
DCGM_EXPORTER_EVIDENCE="DCGM metrics at $DCGM_EXPORTER_URL"
fi
fi

[[ "$PROMETHEUS_DETECTED" == "true" ]] && print_info "Prometheus: detected (port 9090 or systemd active)" || print_detail "Prometheus: not detected"
[[ "$DCGM_EXPORTER_DETECTED" == "true" ]] && print_info "dcgm-exporter: detected (port 9400 or systemd active)" || print_detail "dcgm-exporter: not detected"
[[ "$DCGM_EXPORTER_DETECTED" == "true" ]] && print_info "dcgm-exporter: detected ($DCGM_EXPORTER_EVIDENCE)" || print_detail "dcgm-exporter: not detected"
[[ "$NODE_EXPORTER_DETECTED" == "true" ]] && print_info "node-exporter: detected (port 9100 or systemd active)" || print_detail "node-exporter: not detected"
[[ "$GRAFANA_DETECTED" == "true" ]] && print_info "Grafana: detected (port 3000 or systemd active)" || print_detail "Grafana: not detected"
if [[ "$SLURM_EXPORTER_DETECTED" == "maybe" ]]; then
Expand Down
73 changes: 73 additions & 0 deletions tests/audit/test_monitoring_process_detection.py
Original file line number Diff line number Diff line change
Expand Up @@ -66,3 +66,76 @@ def test_listening_port_detects_dcgm_exporter(collector: str) -> None:
)
assert run.returncode == 0, run.stderr
assert run.stdout.strip() == "dcgm=true"


# Slurm on Kubernetes: the head node is a pod and the exporter is a DaemonSet
# behind a Service, so only the Service probe can see it.
SERVICE_URL = "http://nvidia-dcgm-exporter.gpu-operator.svc:9400/metrics"
POD_RESOLV = "search slurm.svc.cluster.local svc.cluster.local cluster.local\nnameserver 10.96.0.10\n"
HOST_RESOLV = "search example.internal\nnameserver 10.0.0.2\n"
DCGM_METRICS = 'DCGM_FI_DEV_GPU_TEMP{gpu="0"} 41\n'


def service_probe_block() -> str:
return bashtest.extract_block(
WORKLOAD / "cluster-audit-slurm.sh",
"# Monitoring stack detection:",
'[[ "$DCGM_EXPORTER_DETECTED" == "true" ]] && print_info',
)


def run_service_probe(
tmp_path: Path, resolv: str, curl: str, env: dict[str, str] | None = None
) -> bashtest.BashRun:
resolv_conf = tmp_path / "resolv.conf"
resolv_conf.write_text(resolv)
return bashtest.run_bash(
service_probe_block() + REPORT,
stubs={
**PRINT_STUBS,
"ss": "exit 0",
"systemctl": "exit 1",
"pgrep": "exit 1",
"curl": curl,
},
env={"CLUSTERMAX_RESOLV_CONF": str(resolv_conf), **(env or {})},
)


def test_pod_head_node_detects_exporter_service(tmp_path: Path) -> None:
run = run_service_probe(tmp_path, POD_RESOLV, f"printf '{DCGM_METRICS}'")
assert run.returncode == 0, run.stderr
assert run.stdout.strip() == "dcgm=true"
assert run.calls("curl")[0][-1] == SERVICE_URL
assert ["dcgm-exporter: detected (DCGM metrics at " + SERVICE_URL + ")"] in run.calls("print_info")


@pytest.mark.parametrize(
"curl",
["printf 'node_load1 0.5\\n'", "exit 7"],
ids=["not-dcgm-metrics", "unreachable"],
)
def test_service_without_dcgm_metrics_is_not_detected(tmp_path: Path, curl: str) -> None:
run = run_service_probe(tmp_path, POD_RESOLV, curl)
assert run.returncode == 0, run.stderr
assert run.stdout.strip() == "dcgm=false"


def test_bare_metal_head_node_does_not_query_service(tmp_path: Path) -> None:
run = run_service_probe(tmp_path, HOST_RESOLV, f"printf '{DCGM_METRICS}'")
assert run.returncode == 0, run.stderr
assert run.stdout.strip() == "dcgm=false"
assert run.calls("curl") == []


def test_exporter_url_override_is_queried(tmp_path: Path) -> None:
url = "http://dcgm-exporter.monitoring.svc:9400/metrics"
run = run_service_probe(
tmp_path,
HOST_RESOLV,
f"printf '{DCGM_METRICS}'",
env={"CLUSTERMAX_DCGM_EXPORTER_URL": url},
)
assert run.returncode == 0, run.stderr
assert run.stdout.strip() == "dcgm=true"
assert run.calls("curl")[0][-1] == url