diff --git a/.github/docker/build-rocm-sdk-image.sh b/.github/docker/build-rocm-sdk-image.sh new file mode 100755 index 000000000..0c0852743 --- /dev/null +++ b/.github/docker/build-rocm-sdk-image.sh @@ -0,0 +1,489 @@ +#!/usr/bin/env bash +# Shared builder for nightly ROCm runtime images. +# Tarball SDK (Ubuntu / manylinux) or yum (RHEL 8 / RHEL 9) is chosen from --context: +# install-rvs.sh present -> yum (rhel8 / rhel9 from the directory name) +# otherwise -> TheRock SDK tarball +# +# Examples: +# ./build-rocm-sdk-image.sh --context .github/docker/rvs-nightly-rocm-ubuntu22.04 --from-tarball amdrocm10-rvs-...-Linux.tar.gz +# ./build-rocm-sdk-image.sh --context .github/docker/rvs-nightly-rocm-rhel8 --channel nightly +# ./build-rocm-sdk-image.sh --context .github/docker/rvs-nightly-rocm-rhel9 --resolve-only +# +# Nightly listing default: https://nightly.repo.amd.com/rocm/core/tarball/ +# Override with ROCM_SDK_NIGHTLY_BASE_URL / ROCM_SDK_NIGHTLY_INDEX_URL. + +set -euo pipefail + +SELF_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +CONTEXT="" +IMAGE_REPO="${RVS_NIGHTLY_DOCKER_IMAGE:-}" + +ROCM_VERSION="${ROCM_VERSION:-}" +GPU_FAMILY="${GPU_FAMILY:-multiarch}" +CHANNEL="nightly" +IMAGE_TAG="" +ROCM_SDK_BASE_URL="" +FROM_TARBALL="" +FALLBACK_LATEST_SDK="${RVS_DOCKER_SDK_FALLBACK_LATEST:-true}" +GPU_TARGET="${GPU_TARGET:-multiarch}" +RESOLVE_ONLY=false +ROCM_REPO_BASEURL="${ROCM_REPO_BASEURL:-}" +RVS_REPO_BASEURL="${RVS_REPO_BASEURL:-}" +ROCM_GPG_KEY="${ROCM_GPG_KEY:-}" +ROCM_PACKAGE="${ROCM_PACKAGE:-}" +RVS_PACKAGE="${RVS_PACKAGE:-}" +ROCM_MAJOR="" +ROCM_SNAPSHOT="" +RVS_REPO_OVERRIDE=false +RHEL_DIST="" +ROCM_NIGHTLY_INDEX="" +RVS_NIGHTLY_REPO_DEFAULT="" +RVS_STABLE_REPO_DEFAULT="" + +# Override with ROCM_SDK_NIGHTLY_BASE_URL (and optional ROCM_SDK_NIGHTLY_INDEX_URL). +if [ -n "${ROCM_SDK_NIGHTLY_BASE_URL:-}" ]; then + NIGHTLY_BASE="${ROCM_SDK_NIGHTLY_BASE_URL%/}" + NIGHTLY_INDEX="${ROCM_SDK_NIGHTLY_INDEX_URL:-${NIGHTLY_BASE}/}" +elif [ -n "${ROCM_SDK_NIGHTLY_INDEX_URL:-}" ]; then + NIGHTLY_INDEX="${ROCM_SDK_NIGHTLY_INDEX_URL}" + NIGHTLY_BASE="${NIGHTLY_INDEX%/}" +else + NIGHTLY_INDEX="https://nightly.repo.amd.com/rocm/core/tarball/" + NIGHTLY_BASE="https://nightly.repo.amd.com/rocm/core/tarball" +fi +RELEASE_LIST="${ROCM_SDK_RELEASE_URL:-https://repo.amd.com/rocm/tarball/}" +RELEASE_BASE="${ROCM_SDK_RELEASE_BASE_URL:-https://repo.amd.com/rocm/tarball}" + +usage() { + sed -n '2,12p' "$0" + exit 1 +} + +resolve_sdk_base() { + local ver="$1" + if echo "$ver" | grep -qE '^[0-9]+\.[0-9]+\.[0-9]+a[0-9]+'; then + ROCM_SDK_BASE_URL="$NIGHTLY_BASE" + elif echo "$ver" | grep -qE '^[0-9]+\.[0-9]+\.[0-9]+$'; then + ROCM_SDK_BASE_URL="$RELEASE_BASE" + else + echo "::error::Unrecognized ROCm version format: $ver" >&2 + exit 1 + fi +} + +fallback_latest_sdk_enabled() { + case "${FALLBACK_LATEST_SDK}" in + true|1|yes|YES) return 0 ;; + *) return 1 ;; + esac +} + +fetch_latest_nightly_sdk_for_line() { + local listing="$1" + local major="$2" + local minor="$3" + local prefix="${major}.${minor}.0a" + grep -oE "therock-dist-linux-${GPU_FAMILY}-${prefix}[0-9]+" "$listing" \ + | sed "s|^therock-dist-linux-${GPU_FAMILY}-||" | sort -V | tail -1 +} + +fetch_latest_nightly_sdk_for_major() { + local listing="$1" + local major="$2" + grep -oE "therock-dist-linux-${GPU_FAMILY}-${major}\.[0-9]+\.[0-9]+a[0-9]+" "$listing" \ + | sed "s|^therock-dist-linux-${GPU_FAMILY}-||" | sort -V | tail -1 +} + +fetch_latest_nightly_sdk_any() { + local listing="$1" + grep -oE "therock-dist-linux-${GPU_FAMILY}-[0-9]+\.[0-9]+\.[0-9]+a[0-9]+" "$listing" \ + | sed "s|^therock-dist-linux-${GPU_FAMILY}-||" | sort -V | tail -1 +} + +# Same major.minor, else same major (10.0 missing -> newest 10.x), else newest nightly on the listing. +pick_fallback_nightly_sdk() { + local listing="$1" major="$2" minor="$3" exact="$4" ver + ver="$(fetch_latest_nightly_sdk_for_line "$listing" "$major" "$minor")" + if [ -n "$ver" ]; then + echo "::warning::SDK ${exact} missing on ${NIGHTLY_INDEX}; using latest ${major}.${minor}.0a* ${ver}" >&2 + printf '%s\n' "$ver" + return 0 + fi + ver="$(fetch_latest_nightly_sdk_for_major "$listing" "$major")" + if [ -n "$ver" ]; then + echo "::warning::No ${major}.${minor}.0a* SDK for ${exact}; using latest ROCm ${major}.x ${ver}" >&2 + printf '%s\n' "$ver" + return 0 + fi + ver="$(fetch_latest_nightly_sdk_any "$listing")" + if [ -n "$ver" ]; then + echo "::warning::No ROCm ${major}.x SDK for ${exact}; using latest nightly ${ver}" >&2 + printf '%s\n' "$ver" + return 0 + fi + return 1 +} + +fetch_latest_version() { + local mode="$1" + local listing_tmp versions + listing_tmp="$(mktemp)" + if [ "$mode" = "nightly" ]; then + wget -q -O "$listing_tmp" "$NIGHTLY_INDEX" + versions=$(grep -oE "therock-dist-linux-${GPU_FAMILY}-[0-9]+\.[0-9]+\.[0-9]+a[0-9]+" "$listing_tmp" \ + | sed "s|^therock-dist-linux-${GPU_FAMILY}-||" | sort -V | tail -1) + else + wget -q -O "$listing_tmp" "$RELEASE_LIST" + versions=$(grep -oE "therock-dist-linux-${GPU_FAMILY}-[0-9]+\.[0-9]+\.[0-9]+" "$listing_tmp" \ + | sed "s|^therock-dist-linux-${GPU_FAMILY}-||" | sort -V | tail -1) + fi + rm -f "$listing_tmp" + if [ -z "$versions" ]; then + echo "::error::No SDK version found for ${GPU_FAMILY} (${mode})" >&2 + exit 1 + fi + ROCM_VERSION="$versions" +} + +resolve_rocm_from_tarball() { + local name="$1" + local base="${name##*/}" + local major minor build_date exact listing_tmp sdk_file prefix + + if [[ "$base" != *-Linux.tar.gz ]]; then + echo "::error::--from-tarball requires a *-Linux.tar.gz relocatable tarball; got: ${base}" >&2 + exit 1 + fi + + if [[ "$base" =~ -r([0-9]{2})([0-9]{2})\.([0-9]{8})-Linux\.tar\.gz$ ]]; then + major=$((10#${BASH_REMATCH[1]})) + minor=$((10#${BASH_REMATCH[2]})) + build_date="${BASH_REMATCH[3]}" + else + echo "::error::Cannot parse ROCm version from tar tarball (expected ...-rMMmm.yyyymmdd-Linux.tar.gz): ${base}" >&2 + exit 1 + fi + + if [ "$CHANNEL" = "release" ]; then + prefix="${major}.${minor}." + listing_tmp="$(mktemp)" + wget -q -O "$listing_tmp" "$RELEASE_LIST" + ROCM_VERSION=$(grep -oE "therock-dist-linux-${GPU_FAMILY}-${prefix}[0-9]+" "$listing_tmp" \ + | sed "s|^therock-dist-linux-${GPU_FAMILY}-||" | sort -V | tail -1) + rm -f "$listing_tmp" + sdk_file="therock-dist-linux-${GPU_FAMILY}-${ROCM_VERSION}.tar.gz" + resolve_sdk_base "$ROCM_VERSION" + else + exact="${major}.${minor}.0a${build_date}" + sdk_file="therock-dist-linux-${GPU_FAMILY}-${exact}.tar.gz" + listing_tmp="$(mktemp)" + wget -q -O "$listing_tmp" "$NIGHTLY_INDEX" + if grep -qF "$sdk_file" "$listing_tmp"; then + ROCM_VERSION="$exact" + ROCM_SDK_BASE_URL="$NIGHTLY_BASE" + elif fallback_latest_sdk_enabled; then + if ! ROCM_VERSION="$(pick_fallback_nightly_sdk "$listing_tmp" "$major" "$minor" "$exact")"; then + rm -f "$listing_tmp" + echo "::error::No ${GPU_FAMILY} SDK ${exact} for tar ${base} (missing ${sdk_file} on ${NIGHTLY_INDEX}) and no nightly build on that listing" >&2 + exit 1 + fi + ROCM_SDK_BASE_URL="$NIGHTLY_BASE" + else + rm -f "$listing_tmp" + echo "::error::No ${GPU_FAMILY} SDK ${exact} for tar ${base} (missing ${sdk_file} on ${NIGHTLY_INDEX}). Pass --fallback-latest-sdk or set RVS_DOCKER_SDK_FALLBACK_LATEST=true to use the latest nightly build." >&2 + exit 1 + fi + rm -f "$listing_tmp" + fi + + if [ -z "$ROCM_VERSION" ]; then + echo "::error::No ${CHANNEL} SDK found for ROCm ${major}.${minor} (${GPU_FAMILY}) from tar ${base}" >&2 + exit 1 + fi + echo "Resolved ROCm ${ROCM_VERSION} from tar tarball ${base} (r$(printf '%02d%02d' "$major" "$minor").${build_date})" +} + +yum_context() { + [ -f "${CONTEXT}/install-rvs.sh" ] +} + +init_rhel_dist() { + case "$(basename "$CONTEXT")" in + *rhel8*) RHEL_DIST=rhel8 ;; + *rhel9*) RHEL_DIST=rhel9 ;; + *) + echo "::error::Yum image context ${CONTEXT} must be an *rhel8* or *rhel9* docker directory" >&2 + exit 1 + ;; + esac + local idx_var rvs_var gpg_var + idx_var="RVS_NIGHTLY_$(printf '%s' "$RHEL_DIST" | tr '[:lower:]' '[:upper:]')_ROCM_REPO_INDEX" + rvs_var="RVS_NIGHTLY_$(printf '%s' "$RHEL_DIST" | tr '[:lower:]' '[:upper:]')_RVS_REPO_BASEURL" + gpg_var="RVS_NIGHTLY_$(printf '%s' "$RHEL_DIST" | tr '[:lower:]' '[:upper:]')_GPG_KEY" + if [ -z "$ROCM_NIGHTLY_INDEX" ]; then + ROCM_NIGHTLY_INDEX="${!idx_var:-https://nightly.repo.amd.com/rocm/core/packages/${RHEL_DIST}/}" + fi + if [ -z "$RVS_REPO_BASEURL" ]; then + RVS_REPO_BASEURL="${!rvs_var:-}" + fi + if [ -z "$ROCM_GPG_KEY" ]; then + ROCM_GPG_KEY="${!gpg_var:-https://stable.repo.amd.com/rocm/gpg/packages.gpg}" + fi + RVS_NIGHTLY_REPO_DEFAULT="https://nightly.repo.amd.com/rocm/extras/rvs/packages/${RHEL_DIST}/x86_64" + RVS_STABLE_REPO_DEFAULT="https://stable.repo.amd.com/rocm/extras/rvs/packages/${RHEL_DIST}/x86_64" +} + +fetch_url() { + wget -q -O - "$1" 2>/dev/null || curl -fsSL --max-time 60 --retry 2 --retry-delay 2 "$1" +} + +repo_has_metadata() { + local base="${1%/}" + curl -fsSL -o /dev/null --max-time 20 --retry 1 "${base}/repodata/repomd.xml" 2>/dev/null \ + || wget -q -O /dev/null --timeout=20 "${base}/repodata/repomd.xml" 2>/dev/null +} + +resolve_rvs_repo() { + if [ -n "$RVS_REPO_BASEURL" ]; then + if repo_has_metadata "$RVS_REPO_BASEURL"; then + return 0 + fi + if [ "$RVS_REPO_OVERRIDE" = true ]; then + echo "::error::RVS yum repo has no repodata: ${RVS_REPO_BASEURL}" >&2 + exit 1 + fi + echo "::warning::RVS repo ${RVS_REPO_BASEURL} has no repodata; trying defaults" >&2 + RVS_REPO_BASEURL="" + fi + if repo_has_metadata "$RVS_NIGHTLY_REPO_DEFAULT"; then + RVS_REPO_BASEURL="$RVS_NIGHTLY_REPO_DEFAULT" + echo "::notice::Using nightly RVS extras repo" + return 0 + fi + if repo_has_metadata "$RVS_STABLE_REPO_DEFAULT"; then + RVS_REPO_BASEURL="$RVS_STABLE_REPO_DEFAULT" + echo "::warning::Nightly RVS extras yum is unpublished; using stable extras ${RVS_STABLE_REPO_DEFAULT}" >&2 + return 0 + fi + echo "::error::No working RVS yum repo (tried nightly extras and ${RVS_STABLE_REPO_DEFAULT})" >&2 + exit 1 +} + +latest_rocm_snapshot() { + local html + html="$(fetch_url "$ROCM_NIGHTLY_INDEX")" + printf '%s' "$html" | grep -oE '[0-9]{8}-[0-9]+' | sort -u | tail -n 1 +} + +gpu_target_is_multiarch() { + case "${GPU_TARGET}" in + multiarch|all|"") return 0 ;; + *) return 1 ;; + esac +} + +latest_rocm_package_from_listing() { + local listing pkg + listing="$(fetch_url "${ROCM_REPO_BASEURL%/}/")" + if gpu_target_is_multiarch; then + pkg="$(printf '%s' "$listing" \ + | grep -oE 'amdrocm[0-9]+\.[0-9]+-[0-9]+\.[0-9]+\.[0-9]+~[0-9A-Za-z._~+-]+\.x86_64\.rpm' \ + | sed 's/-[0-9][0-9]*\.[0-9].*//' \ + | sort -uV | tail -n 1 || true)" + else + pkg="$(printf '%s' "$listing" \ + | grep -oE "amdrocm[0-9.]+-${GPU_TARGET}-[0-9A-Za-z._~+-]+\.x86_64\.rpm" \ + | sed 's/-[0-9][0-9A-Za-z._~+-]*\.x86_64\.rpm$//' \ + | sort -uV | tail -n 1 || true)" + fi + printf '%s\n' "$pkg" +} + +emit_github_yum() { + if [ -z "${GITHUB_OUTPUT:-}" ]; then + return 0 + fi + { + echo "rocm_snapshot=${ROCM_SNAPSHOT}" + echo "rocm_version=${ROCM_VERSION}" + echo "rocm_package=${ROCM_PACKAGE}" + echo "rocm_major=${ROCM_MAJOR}" + echo "rocm_repo_baseurl=${ROCM_REPO_BASEURL}" + echo "rvs_repo_baseurl=${RVS_REPO_BASEURL}" + echo "rvs_package=${RVS_PACKAGE}" + echo "gpu_target=${GPU_TARGET}" + echo "rocm_install_path=${ROCM_INSTALL_PATH:-/opt/rocm}" + echo "tarball_name=amdrocm${ROCM_MAJOR}-rvs-nightly-${RHEL_DIST}" + echo "tarball_url=${ROCM_REPO_BASEURL}" + } >> "$GITHUB_OUTPUT" +} + +resolve_nightly_repos() { + if [ "$CHANNEL" != "nightly" ]; then + echo "::error::RHEL docker tests only support --channel nightly (got ${CHANNEL})" >&2 + exit 1 + fi + + if [ -z "$ROCM_REPO_BASEURL" ]; then + ROCM_SNAPSHOT="$(latest_rocm_snapshot)" + if [ -z "$ROCM_SNAPSHOT" ]; then + echo "::error::Could not find a nightly ROCm snapshot under ${ROCM_NIGHTLY_INDEX}" >&2 + exit 1 + fi + ROCM_REPO_BASEURL="${ROCM_NIGHTLY_INDEX%/}/${ROCM_SNAPSHOT}/x86_64" + else + ROCM_SNAPSHOT="$(printf '%s' "$ROCM_REPO_BASEURL" | grep -oE '[0-9]{8}-[0-9]+' | tail -n 1 || true)" + fi + + if [ -z "$ROCM_PACKAGE" ]; then + ROCM_PACKAGE="$(latest_rocm_package_from_listing)" + fi + if [ -z "$ROCM_PACKAGE" ]; then + echo "::warning::Could not scrape package name from ${ROCM_REPO_BASEURL}; docker build will install amdrocm*-${GPU_TARGET}" >&2 + ROCM_PACKAGE="" + ROCM_MAJOR="${ROCM_MAJOR:-10}" + ROCM_VERSION="${ROCM_VERSION:-${ROCM_SNAPSHOT:-nightly}}" + else + if [[ "$ROCM_PACKAGE" =~ ^amdrocm([0-9]+) ]]; then + ROCM_MAJOR="${BASH_REMATCH[1]}" + else + echo "::error::Cannot parse ROCm major from package ${ROCM_PACKAGE}" >&2 + exit 1 + fi + ROCM_VERSION="${ROCM_VERSION:-${ROCM_SNAPSHOT:-$ROCM_PACKAGE}}" + fi + + if [ -z "$RVS_PACKAGE" ]; then + RVS_PACKAGE="amdrocm${ROCM_MAJOR}-rvs" + fi + + resolve_rvs_repo + + if [[ "${ROCM_PACKAGE}" =~ ^amdrocm([0-9]+\.[0-9]+) ]]; then + ROCM_INSTALL_PATH="${ROCM_INSTALL_PATH:-/opt/rocm/core-${BASH_REMATCH[1]}}" + else + ROCM_INSTALL_PATH="${ROCM_INSTALL_PATH:-/opt/rocm}" + fi + + echo "Resolved ${RHEL_DIST} nightly repos" + echo " ROCm snapshot : ${ROCM_SNAPSHOT:-n/a}" + echo " ROCm repo : ${ROCM_REPO_BASEURL}" + echo " ROCm package : ${ROCM_PACKAGE:-amdrocm*-${GPU_TARGET}}" + echo " ROCm major : ${ROCM_MAJOR}" + echo " RVS repo : ${RVS_REPO_BASEURL}" + echo " RVS package : ${RVS_PACKAGE}" + echo " GPU target : ${GPU_TARGET}" + echo " ROCm path : ${ROCM_INSTALL_PATH:-/opt/rocm}" + emit_github_yum +} + +build_yum_image() { + local build_args + IMAGE_TAG="${IMAGE_TAG:-${IMAGE_REPO}:${ROCM_VERSION}}" + echo "Building docker image ${IMAGE_TAG}" + echo " Context : ${CONTEXT}" + echo " Base image : $(awk '/^FROM / { print $2; exit }' "${CONTEXT}/Dockerfile")" + echo " ROCm version : ${ROCM_VERSION}" + ROCM_INSTALL_PATH="${ROCM_INSTALL_PATH:-/opt/rocm}" + build_args=( + --build-arg "ROCM_VERSION=${ROCM_VERSION}" + --build-arg "ROCM_REPO_BASEURL=${ROCM_REPO_BASEURL}" + --build-arg "RVS_REPO_BASEURL=${RVS_REPO_BASEURL}" + --build-arg "ROCM_GPG_KEY=${ROCM_GPG_KEY}" + --build-arg "GPU_TARGET=${GPU_TARGET}" + --build-arg "ROCM_INSTALL_PATH=${ROCM_INSTALL_PATH}" + ) + if [ -n "$ROCM_PACKAGE" ]; then + build_args+=(--build-arg "ROCM_PACKAGE=${ROCM_PACKAGE}") + fi + if [ -n "$RVS_PACKAGE" ]; then + build_args+=(--build-arg "RVS_PACKAGE=${RVS_PACKAGE}") + fi + docker build -f "${CONTEXT}/Dockerfile" "${build_args[@]}" -t "${IMAGE_TAG}" "${CONTEXT}" + docker tag "${IMAGE_TAG}" "${IMAGE_REPO}:latest" + echo "::notice::Tagged ${IMAGE_TAG} and ${IMAGE_REPO}:latest" +} + +build_tarball_image() { + if [ -n "$FROM_TARBALL" ]; then + resolve_rocm_from_tarball "$(basename "$FROM_TARBALL")" + elif [ -z "$ROCM_VERSION" ]; then + fetch_latest_version "$CHANNEL" + fi + resolve_sdk_base "$ROCM_VERSION" + IMAGE_TAG="${IMAGE_TAG:-${IMAGE_REPO}:${ROCM_VERSION}}" + echo "Building docker image ${IMAGE_TAG}" + echo " ROCm version : ${ROCM_VERSION}" + echo " GPU family : ${GPU_FAMILY}" + echo " SDK base URL : ${ROCM_SDK_BASE_URL}" + echo " Context : ${CONTEXT}" + echo " Base image : $(awk '/^FROM / { print $2; exit }' "${CONTEXT}/Dockerfile")" + ROCM_INSTALL_PATH="${ROCM_INSTALL_PATH:-/opt/rocm/install}" + docker build \ + -f "${CONTEXT}/Dockerfile" \ + --build-arg "ROCM_VERSION=${ROCM_VERSION}" \ + --build-arg "GPU_FAMILY=${GPU_FAMILY}" \ + --build-arg "ROCM_SDK_BASE_URL=${ROCM_SDK_BASE_URL}" \ + --build-arg "ROCM_INSTALL_PATH=${ROCM_INSTALL_PATH}" \ + -t "${IMAGE_TAG}" \ + "${CONTEXT}" + docker tag "${IMAGE_TAG}" "${IMAGE_REPO}:latest" + echo "::notice::Tagged ${IMAGE_TAG} and ${IMAGE_REPO}:latest" +} + +while [ $# -gt 0 ]; do + case "$1" in + --context) CONTEXT="$2"; shift 2 ;; + --rocm-version) ROCM_VERSION="$2"; shift 2 ;; + --from-tarball) FROM_TARBALL="$2"; shift 2 ;; + --gpu-family) GPU_FAMILY="$2"; GPU_TARGET="$2"; shift 2 ;; + --gpu-target) GPU_TARGET="$2"; GPU_FAMILY="$2"; shift 2 ;; + --channel) CHANNEL="$2"; shift 2 ;; + --tag) IMAGE_TAG="$2"; shift 2 ;; + --rocm-repo) ROCM_REPO_BASEURL="$2"; shift 2 ;; + --rvs-repo) RVS_REPO_BASEURL="$2"; RVS_REPO_OVERRIDE=true; shift 2 ;; + --rocm-package) ROCM_PACKAGE="$2"; shift 2 ;; + --rvs-package) RVS_PACKAGE="$2"; shift 2 ;; + --resolve-only) RESOLVE_ONLY=true; shift ;; + --fallback-latest-sdk) FALLBACK_LATEST_SDK=true; shift ;; + -h|--help) usage ;; + *) echo "Unknown arg: $1" >&2; usage ;; + esac +done + +if [ -z "$CONTEXT" ]; then + if [ -f "${SELF_DIR}/Dockerfile" ]; then + CONTEXT="$SELF_DIR" + else + echo "::error::Pass --context (directory with Dockerfile), or place this script next to the Dockerfile." >&2 + exit 1 + fi +fi +CONTEXT="$(cd "$CONTEXT" && pwd)" +if [ ! -f "${CONTEXT}/Dockerfile" ]; then + echo "::error::No Dockerfile in context: ${CONTEXT}" >&2 + exit 1 +fi + +if [ -z "$IMAGE_REPO" ]; then + IMAGE_REPO="$(basename "$CONTEXT"):latest" +fi +IMAGE_REPO="${IMAGE_REPO%%:*}" + +if yum_context; then + init_rhel_dist + if [ -n "$FROM_TARBALL" ]; then + echo "::notice::--from-tarball ${FROM_TARBALL} ignored; ${RHEL_DIST} image uses nightly dnf repos" + fi + resolve_nightly_repos + if [ "$RESOLVE_ONLY" = true ]; then + exit 0 + fi + build_yum_image +else + if [ "$RESOLVE_ONLY" = true ]; then + echo "::error::--resolve-only is only supported for RHEL yum images" >&2 + exit 1 + fi + build_tarball_image +fi diff --git a/.github/docker/rvs-nightly-rocm-manylinux_2_28/Dockerfile b/.github/docker/rvs-nightly-rocm-manylinux_2_28/Dockerfile new file mode 100644 index 000000000..b407df07a --- /dev/null +++ b/.github/docker/rvs-nightly-rocm-manylinux_2_28/Dockerfile @@ -0,0 +1,37 @@ +# ROCm nightly-test runtime image on manylinux_2_28 (AlmaLinux 8). +# Same SDK tarball layout as build_packages_local.sh +# (therock-dist-linux--.tar.gz from multiarch or release listing). +# +# Build on any docker host (typically the self-hosted runner): +# ./.github/docker/build-rocm-sdk-image.sh --context .github/docker/rvs-nightly-rocm-manylinux_2_28 \ +# --rocm-version 10.1.0a20260819 --gpu-family multiarch + +FROM quay.io/pypa/manylinux_2_28_x86_64 + +ARG ROCM_VERSION +ARG GPU_FAMILY=multiarch +ARG ROCM_SDK_BASE_URL +ARG ROCM_INSTALL_PATH=/opt/rocm/install + +RUN yum install -y wget tar pciutils kmod python3 ca-certificates \ + && yum clean all + +RUN test -n "${ROCM_VERSION}" && test -n "${ROCM_SDK_BASE_URL}" + +RUN mkdir -p "${ROCM_INSTALL_PATH}" \ + && wget -q -O /tmp/rocm-sdk.tar.gz \ + "${ROCM_SDK_BASE_URL%/}/therock-dist-linux-${GPU_FAMILY}-${ROCM_VERSION}.tar.gz" \ + && tar -xzf /tmp/rocm-sdk.tar.gz -C "${ROCM_INSTALL_PATH}" --strip-components=1 \ + && rm -f /tmp/rocm-sdk.tar.gz \ + && test -x "${ROCM_INSTALL_PATH}/bin/rocminfo" \ + && test -x "${ROCM_INSTALL_PATH}/bin/amd-smi" + +COPY env-rocm.sh /etc/profile.d/rocm-env.sh + +ENV ROCM_PATH=${ROCM_INSTALL_PATH} +ENV TARGET_ROCM_PATH=${ROCM_INSTALL_PATH} +ENV ROCM_VERSION=${ROCM_VERSION} +ENV GPU_FAMILY=${GPU_FAMILY} +ENV LD_LIBRARY_PATH=${ROCM_INSTALL_PATH}/lib/rocm_sysdeps/lib:${ROCM_INSTALL_PATH}/lib/llvm/lib:${ROCM_INSTALL_PATH}/lib + +WORKDIR /workspace diff --git a/.github/docker/rvs-nightly-rocm-manylinux_2_28/env-rocm.sh b/.github/docker/rvs-nightly-rocm-manylinux_2_28/env-rocm.sh new file mode 100755 index 000000000..d18dfb416 --- /dev/null +++ b/.github/docker/rvs-nightly-rocm-manylinux_2_28/env-rocm.sh @@ -0,0 +1,6 @@ +# Source in container shells: sets ROCm runtime paths (TheRock multiarch layout). +ROCM_PATH="${ROCM_PATH:-/opt/rocm/install}" +TARGET_ROCM_PATH="${TARGET_ROCM_PATH:-$ROCM_PATH}" +export ROCM_PATH TARGET_ROCM_PATH +export PATH="${ROCM_PATH}/bin:${PATH}" +export LD_LIBRARY_PATH="${ROCM_PATH}/lib/rocm_sysdeps/lib:${ROCM_PATH}/lib/llvm/lib:${ROCM_PATH}/lib:${LD_LIBRARY_PATH:-}" diff --git a/.github/docker/rvs-nightly-rocm-manylinux_2_28/setup-on-runner.sh b/.github/docker/rvs-nightly-rocm-manylinux_2_28/setup-on-runner.sh new file mode 100755 index 000000000..853f6ae64 --- /dev/null +++ b/.github/docker/rvs-nightly-rocm-manylinux_2_28/setup-on-runner.sh @@ -0,0 +1,11 @@ +#!/usr/bin/env bash +# Run on the self-hosted GPU runner host to build the manylinux_2_28 ROCm-matched docker image. +# +# cd ROCmValidationSuite +# ./.github/docker/rvs-nightly-rocm-manylinux_2_28/setup-on-runner.sh --from-tarball amdrocm10-rvs-....tar.gz + +set -euo pipefail +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../../.." && pwd)" +cd "$ROOT" +chmod +x .github/docker/build-rocm-sdk-image.sh +exec .github/docker/build-rocm-sdk-image.sh --context .github/docker/rvs-nightly-rocm-manylinux_2_28 "$@" diff --git a/.github/docker/rvs-nightly-rocm-rhel8/Dockerfile b/.github/docker/rvs-nightly-rocm-rhel8/Dockerfile new file mode 100644 index 000000000..d5bbc0e39 --- /dev/null +++ b/.github/docker/rvs-nightly-rocm-rhel8/Dockerfile @@ -0,0 +1,71 @@ +# ROCm + RVS runtime image on RHEL 8 (Rocky Linux 8). +# ROCm comes from nightly yum; RVS from extras yum (nightly if published, else stable). +# +# Build: +# ./.github/docker/build-rocm-sdk-image.sh --context .github/docker/rvs-nightly-rocm-rhel8 --channel nightly + +FROM rockylinux:8 + +ARG ROCM_VERSION +ARG ROCM_REPO_BASEURL +ARG RVS_REPO_BASEURL +ARG ROCM_GPG_KEY=https://stable.repo.amd.com/rocm/gpg/packages.gpg +ARG GPU_TARGET=multiarch +ARG ROCM_PACKAGE +ARG RVS_PACKAGE +ARG ROCM_INSTALL_PATH=/opt/rocm + +RUN dnf -y update \ + && dnf -y install sudo wget ca-certificates dnf-plugins-core \ + && dnf clean all + +RUN test -n "${ROCM_REPO_BASEURL}" && test -n "${RVS_REPO_BASEURL}" + +# Nightly ROCm RPMs are unsigned; gpgcheck=0 is required. +# Default GPU_TARGET=multiarch installs the ISA-less meta (e.g. amdrocm10.1). +# Set GPU_TARGET=gfx942 (etc.) for a single-ISA package. +RUN printf '%s\n' \ + '[amdrocm-nightly]' \ + 'name=ROCm Nightly' \ + "baseurl=${ROCM_REPO_BASEURL}" \ + 'enabled=1' \ + 'gpgcheck=0' \ + "gpgkey=${ROCM_GPG_KEY}" \ + | tee /etc/yum.repos.d/amdrocm-nightly.repo \ + && dnf clean all \ + && if [ -n "${ROCM_PACKAGE}" ]; then \ + dnf -y --nogpgcheck install "${ROCM_PACKAGE}"; \ + elif [ "${GPU_TARGET}" = "multiarch" ] || [ "${GPU_TARGET}" = "all" ]; then \ + pkg="$(dnf repoquery --latest-only --qf '%{name}' 'amdrocm*.*' | grep -E '^amdrocm[0-9]+\.[0-9]+$' | sort -V | tail -n 1)"; \ + test -n "${pkg}"; \ + dnf -y --nogpgcheck install "${pkg}"; \ + else \ + pkg="$(dnf repoquery --latest-only --qf '%{name}' "amdrocm*-${GPU_TARGET}" | grep -E "^amdrocm[0-9.]+-${GPU_TARGET}$" | sort -V | tail -n 1)"; \ + test -n "${pkg}"; \ + dnf -y --nogpgcheck install "${pkg}"; \ + fi \ + && dnf clean all + +COPY install-rvs.sh /tmp/install-rvs.sh +RUN chmod +x /tmp/install-rvs.sh \ + && RVS_REPO_BASEURL="${RVS_REPO_BASEURL}" \ + RVS_PACKAGE="${RVS_PACKAGE}" \ + ROCM_GPG_KEY="${ROCM_GPG_KEY}" \ + /tmp/install-rvs.sh \ + && rm -f /tmp/install-rvs.sh + +COPY env-rocm.sh /etc/profile.d/rocm-env.sh + +# rocminfo lives in /opt/rocm/core-/bin after nightly RPM install. +RUN bash -lc 'source /etc/profile.d/rocm-env.sh; \ + test -x "${ROCM_PATH}/bin/rocminfo"; \ + test -n "${EXTRAS_PATH}" && test -x "${EXTRAS_PATH}/bin/rvs"' + +ENV ROCM_PATH=${ROCM_INSTALL_PATH} +ENV TARGET_ROCM_PATH=${ROCM_INSTALL_PATH} +ENV ROCM_VERSION=${ROCM_VERSION} +ENV GPU_TARGET=${GPU_TARGET} +ENV RHEL_VERSION=8 +ENV PATH=${ROCM_INSTALL_PATH}/bin:${PATH} + +WORKDIR /workspace diff --git a/.github/docker/rvs-nightly-rocm-rhel8/env-rocm.sh b/.github/docker/rvs-nightly-rocm-rhel8/env-rocm.sh new file mode 100644 index 000000000..d5bd3a867 --- /dev/null +++ b/.github/docker/rvs-nightly-rocm-rhel8/env-rocm.sh @@ -0,0 +1,29 @@ +# Source in container shells: RPM-installed ROCm Core + RVS extras. +# Nightly amdrocm RPMs land under /opt/rocm/core-, not /opt/rocm/install. +if [ -z "${ROCM_PATH:-}" ] || [ ! -x "${ROCM_PATH}/bin/rocminfo" ]; then + if [ -x /opt/rocm/bin/rocminfo ]; then + ROCM_PATH=/opt/rocm + else + for _core in /opt/rocm/core-*; do + if [ -x "${_core}/bin/rocminfo" ]; then + ROCM_PATH="${_core}" + break + fi + done + unset _core + fi +fi +ROCM_PATH="${ROCM_PATH:-/opt/rocm}" +TARGET_ROCM_PATH="${TARGET_ROCM_PATH:-$ROCM_PATH}" +if [ -z "${EXTRAS_PATH:-}" ]; then + for _extras in /opt/rocm/extras-*; do + if [ -d "$_extras" ]; then + EXTRAS_PATH="$_extras" + fi + done + unset _extras +fi +EXTRAS_PATH="${EXTRAS_PATH:-}" +export ROCM_PATH TARGET_ROCM_PATH EXTRAS_PATH +export PATH="${EXTRAS_PATH:+${EXTRAS_PATH}/bin:}${ROCM_PATH}/bin:${PATH}" +export LD_LIBRARY_PATH="${EXTRAS_PATH:+${EXTRAS_PATH}/lib:}${ROCM_PATH}/lib:${ROCM_PATH}/lib/llvm/lib:${ROCM_PATH}/lib/llvm/lib/x86_64-unknown-linux-gnu:${ROCM_PATH}/lib/rocm_sysdeps/lib:${LD_LIBRARY_PATH:-}" diff --git a/.github/docker/rvs-nightly-rocm-rhel8/install-rvs.sh b/.github/docker/rvs-nightly-rocm-rhel8/install-rvs.sh new file mode 100755 index 000000000..ca7d5e262 --- /dev/null +++ b/.github/docker/rvs-nightly-rocm-rhel8/install-rvs.sh @@ -0,0 +1,44 @@ +#!/usr/bin/env bash +# Install RVS RPM without resolving unversioned amdrocm-* deps from the +# nightly ROCm repo (those metas pull every GPU ISA). +set -euo pipefail + +RVS_REPO_BASEURL="${RVS_REPO_BASEURL:?}" +RVS_PACKAGE="${RVS_PACKAGE:-amdrocm-rvs}" +ROCM_GPG_KEY="${ROCM_GPG_KEY:-https://stable.repo.amd.com/rocm/gpg/packages.gpg}" + +printf '%s\n' \ + '[rvs]' \ + 'name=ROCm Validation Suite' \ + "baseurl=${RVS_REPO_BASEURL}" \ + 'enabled=1' \ + 'gpgcheck=0' \ + "gpgkey=${ROCM_GPG_KEY}" \ + 'priority=50' \ + | tee /etc/yum.repos.d/rvs.repo + +dnf clean all +mkdir -p /tmp/rvs-rpm +# Nightly ROCm must stay disabled: amdrocm10-rvs Requires: (amdrocm-blas or rocblas) +# and the unversioned amdrocm-blas meta pulls every gfx* package. +dnf -y download --nogpgcheck \ + --disablerepo='*' --enablerepo=rvs \ + --destdir /tmp/rvs-rpm \ + "${RVS_PACKAGE}" +rpm -ivh --nodeps --replacefiles /tmp/rvs-rpm/*.rpm +rm -rf /tmp/rvs-rpm + +test -x /opt/rocm/extras-*/bin/rvs 2>/dev/null || true +rvs_bin="" +for f in /opt/rocm/extras-*/bin/rvs; do + if [ -x "$f" ]; then + rvs_bin="$f" + break + fi +done +if [ -z "$rvs_bin" ]; then + echo "rvs binary missing after RPM install" >&2 + rpm -ql "${RVS_PACKAGE}" 2>/dev/null | head >&2 || true + exit 1 +fi +echo "Installed ${rvs_bin}" diff --git a/.github/docker/rvs-nightly-rocm-rhel8/setup-on-runner.sh b/.github/docker/rvs-nightly-rocm-rhel8/setup-on-runner.sh new file mode 100755 index 000000000..4cfc7588d --- /dev/null +++ b/.github/docker/rvs-nightly-rocm-rhel8/setup-on-runner.sh @@ -0,0 +1,11 @@ +#!/usr/bin/env bash +# Run on the self-hosted GPU runner host to build the RHEL 8 ROCm+RVS docker image. +# +# cd ROCmValidationSuite +# ./.github/docker/rvs-nightly-rocm-rhel8/setup-on-runner.sh --channel nightly + +set -euo pipefail +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../../.." && pwd)" +cd "$ROOT" +chmod +x .github/docker/build-rocm-sdk-image.sh +exec .github/docker/build-rocm-sdk-image.sh --context .github/docker/rvs-nightly-rocm-rhel8 "$@" diff --git a/.github/docker/rvs-nightly-rocm-rhel9/Dockerfile b/.github/docker/rvs-nightly-rocm-rhel9/Dockerfile new file mode 100644 index 000000000..e7fa25a72 --- /dev/null +++ b/.github/docker/rvs-nightly-rocm-rhel9/Dockerfile @@ -0,0 +1,71 @@ +# ROCm + RVS runtime image on RHEL 9 (redhat/ubi9). +# ROCm comes from nightly yum; RVS from extras yum (nightly if published, else stable). +# +# Build: +# ./.github/docker/build-rocm-sdk-image.sh --context .github/docker/rvs-nightly-rocm-rhel9 --channel nightly + +FROM redhat/ubi9 + +ARG ROCM_VERSION +ARG ROCM_REPO_BASEURL +ARG RVS_REPO_BASEURL +ARG ROCM_GPG_KEY=https://stable.repo.amd.com/rocm/gpg/packages.gpg +ARG GPU_TARGET=multiarch +ARG ROCM_PACKAGE +ARG RVS_PACKAGE +ARG ROCM_INSTALL_PATH=/opt/rocm + +RUN dnf -y update \ + && dnf -y install sudo wget ca-certificates dnf-plugins-core \ + && dnf clean all + +RUN test -n "${ROCM_REPO_BASEURL}" && test -n "${RVS_REPO_BASEURL}" + +# Nightly ROCm RPMs are unsigned; gpgcheck=0 is required. +# Default GPU_TARGET=multiarch installs the ISA-less meta (e.g. amdrocm10.1). +# Set GPU_TARGET=gfx942 (etc.) for a single-ISA package. +RUN printf '%s\n' \ + '[amdrocm-nightly]' \ + 'name=ROCm Nightly' \ + "baseurl=${ROCM_REPO_BASEURL}" \ + 'enabled=1' \ + 'gpgcheck=0' \ + "gpgkey=${ROCM_GPG_KEY}" \ + | tee /etc/yum.repos.d/amdrocm-nightly.repo \ + && dnf clean all \ + && if [ -n "${ROCM_PACKAGE}" ]; then \ + dnf -y --nogpgcheck install "${ROCM_PACKAGE}"; \ + elif [ "${GPU_TARGET}" = "multiarch" ] || [ "${GPU_TARGET}" = "all" ]; then \ + pkg="$(dnf repoquery --latest-only --qf '%{name}' 'amdrocm*.*' | grep -E '^amdrocm[0-9]+\.[0-9]+$' | sort -V | tail -n 1)"; \ + test -n "${pkg}"; \ + dnf -y --nogpgcheck install "${pkg}"; \ + else \ + pkg="$(dnf repoquery --latest-only --qf '%{name}' "amdrocm*-${GPU_TARGET}" | grep -E "^amdrocm[0-9.]+-${GPU_TARGET}$" | sort -V | tail -n 1)"; \ + test -n "${pkg}"; \ + dnf -y --nogpgcheck install "${pkg}"; \ + fi \ + && dnf clean all + +COPY install-rvs.sh /tmp/install-rvs.sh +RUN chmod +x /tmp/install-rvs.sh \ + && RVS_REPO_BASEURL="${RVS_REPO_BASEURL}" \ + RVS_PACKAGE="${RVS_PACKAGE}" \ + ROCM_GPG_KEY="${ROCM_GPG_KEY}" \ + /tmp/install-rvs.sh \ + && rm -f /tmp/install-rvs.sh + +COPY env-rocm.sh /etc/profile.d/rocm-env.sh + +# rocminfo lives in /opt/rocm/core-/bin after nightly RPM install. +RUN bash -lc 'source /etc/profile.d/rocm-env.sh; \ + test -x "${ROCM_PATH}/bin/rocminfo"; \ + test -n "${EXTRAS_PATH}" && test -x "${EXTRAS_PATH}/bin/rvs"' + +ENV ROCM_PATH=${ROCM_INSTALL_PATH} +ENV TARGET_ROCM_PATH=${ROCM_INSTALL_PATH} +ENV ROCM_VERSION=${ROCM_VERSION} +ENV GPU_TARGET=${GPU_TARGET} +ENV RHEL_VERSION=9 +ENV PATH=${ROCM_INSTALL_PATH}/bin:${PATH} + +WORKDIR /workspace diff --git a/.github/docker/rvs-nightly-rocm-rhel9/env-rocm.sh b/.github/docker/rvs-nightly-rocm-rhel9/env-rocm.sh new file mode 100644 index 000000000..d5bd3a867 --- /dev/null +++ b/.github/docker/rvs-nightly-rocm-rhel9/env-rocm.sh @@ -0,0 +1,29 @@ +# Source in container shells: RPM-installed ROCm Core + RVS extras. +# Nightly amdrocm RPMs land under /opt/rocm/core-, not /opt/rocm/install. +if [ -z "${ROCM_PATH:-}" ] || [ ! -x "${ROCM_PATH}/bin/rocminfo" ]; then + if [ -x /opt/rocm/bin/rocminfo ]; then + ROCM_PATH=/opt/rocm + else + for _core in /opt/rocm/core-*; do + if [ -x "${_core}/bin/rocminfo" ]; then + ROCM_PATH="${_core}" + break + fi + done + unset _core + fi +fi +ROCM_PATH="${ROCM_PATH:-/opt/rocm}" +TARGET_ROCM_PATH="${TARGET_ROCM_PATH:-$ROCM_PATH}" +if [ -z "${EXTRAS_PATH:-}" ]; then + for _extras in /opt/rocm/extras-*; do + if [ -d "$_extras" ]; then + EXTRAS_PATH="$_extras" + fi + done + unset _extras +fi +EXTRAS_PATH="${EXTRAS_PATH:-}" +export ROCM_PATH TARGET_ROCM_PATH EXTRAS_PATH +export PATH="${EXTRAS_PATH:+${EXTRAS_PATH}/bin:}${ROCM_PATH}/bin:${PATH}" +export LD_LIBRARY_PATH="${EXTRAS_PATH:+${EXTRAS_PATH}/lib:}${ROCM_PATH}/lib:${ROCM_PATH}/lib/llvm/lib:${ROCM_PATH}/lib/llvm/lib/x86_64-unknown-linux-gnu:${ROCM_PATH}/lib/rocm_sysdeps/lib:${LD_LIBRARY_PATH:-}" diff --git a/.github/docker/rvs-nightly-rocm-rhel9/install-rvs.sh b/.github/docker/rvs-nightly-rocm-rhel9/install-rvs.sh new file mode 100755 index 000000000..ca7d5e262 --- /dev/null +++ b/.github/docker/rvs-nightly-rocm-rhel9/install-rvs.sh @@ -0,0 +1,44 @@ +#!/usr/bin/env bash +# Install RVS RPM without resolving unversioned amdrocm-* deps from the +# nightly ROCm repo (those metas pull every GPU ISA). +set -euo pipefail + +RVS_REPO_BASEURL="${RVS_REPO_BASEURL:?}" +RVS_PACKAGE="${RVS_PACKAGE:-amdrocm-rvs}" +ROCM_GPG_KEY="${ROCM_GPG_KEY:-https://stable.repo.amd.com/rocm/gpg/packages.gpg}" + +printf '%s\n' \ + '[rvs]' \ + 'name=ROCm Validation Suite' \ + "baseurl=${RVS_REPO_BASEURL}" \ + 'enabled=1' \ + 'gpgcheck=0' \ + "gpgkey=${ROCM_GPG_KEY}" \ + 'priority=50' \ + | tee /etc/yum.repos.d/rvs.repo + +dnf clean all +mkdir -p /tmp/rvs-rpm +# Nightly ROCm must stay disabled: amdrocm10-rvs Requires: (amdrocm-blas or rocblas) +# and the unversioned amdrocm-blas meta pulls every gfx* package. +dnf -y download --nogpgcheck \ + --disablerepo='*' --enablerepo=rvs \ + --destdir /tmp/rvs-rpm \ + "${RVS_PACKAGE}" +rpm -ivh --nodeps --replacefiles /tmp/rvs-rpm/*.rpm +rm -rf /tmp/rvs-rpm + +test -x /opt/rocm/extras-*/bin/rvs 2>/dev/null || true +rvs_bin="" +for f in /opt/rocm/extras-*/bin/rvs; do + if [ -x "$f" ]; then + rvs_bin="$f" + break + fi +done +if [ -z "$rvs_bin" ]; then + echo "rvs binary missing after RPM install" >&2 + rpm -ql "${RVS_PACKAGE}" 2>/dev/null | head >&2 || true + exit 1 +fi +echo "Installed ${rvs_bin}" diff --git a/.github/docker/rvs-nightly-rocm-rhel9/setup-on-runner.sh b/.github/docker/rvs-nightly-rocm-rhel9/setup-on-runner.sh new file mode 100755 index 000000000..ccc941af3 --- /dev/null +++ b/.github/docker/rvs-nightly-rocm-rhel9/setup-on-runner.sh @@ -0,0 +1,11 @@ +#!/usr/bin/env bash +# Run on the self-hosted GPU runner host to build the RHEL 9 ROCm+RVS docker image. +# +# cd ROCmValidationSuite +# ./.github/docker/rvs-nightly-rocm-rhel9/setup-on-runner.sh --channel nightly + +set -euo pipefail +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../../.." && pwd)" +cd "$ROOT" +chmod +x .github/docker/build-rocm-sdk-image.sh +exec .github/docker/build-rocm-sdk-image.sh --context .github/docker/rvs-nightly-rocm-rhel9 "$@" diff --git a/.github/docker/rvs-nightly-rocm-ubuntu22.04/Dockerfile b/.github/docker/rvs-nightly-rocm-ubuntu22.04/Dockerfile new file mode 100644 index 000000000..5275c0b2c --- /dev/null +++ b/.github/docker/rvs-nightly-rocm-ubuntu22.04/Dockerfile @@ -0,0 +1,43 @@ +# ROCm nightly-test runtime image on Ubuntu 22.04 — same SDK tarball layout as build_packages_local.sh +# (therock-dist-linux--.tar.gz from multiarch or release listing). +# +# Build on any docker host (typically the self-hosted runner): +# ./.github/docker/build-rocm-sdk-image.sh --context .github/docker/rvs-nightly-rocm-ubuntu22.04 \ +# --from-tarball amdrocm10-rvs-...-Linux.tar.gz + +FROM ubuntu:22.04 + +ARG ROCM_VERSION +ARG GPU_FAMILY=multiarch +ARG ROCM_SDK_BASE_URL +ARG ROCM_INSTALL_PATH=/opt/rocm/install + +RUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends \ + ca-certificates \ + wget \ + tar \ + pciutils \ + kmod \ + python3 \ + && rm -rf /var/lib/apt/lists/* + +RUN test -n "${ROCM_VERSION}" && test -n "${ROCM_SDK_BASE_URL}" + +RUN mkdir -p "${ROCM_INSTALL_PATH}" \ + && wget -q -O /tmp/rocm-sdk.tar.gz \ + "${ROCM_SDK_BASE_URL%/}/therock-dist-linux-${GPU_FAMILY}-${ROCM_VERSION}.tar.gz" \ + && tar -xzf /tmp/rocm-sdk.tar.gz -C "${ROCM_INSTALL_PATH}" --strip-components=1 \ + && rm -f /tmp/rocm-sdk.tar.gz \ + && test -x "${ROCM_INSTALL_PATH}/bin/rocminfo" \ + && test -x "${ROCM_INSTALL_PATH}/bin/amd-smi" + +COPY env-rocm.sh /etc/profile.d/rocm-env.sh + +ENV ROCM_PATH=${ROCM_INSTALL_PATH} +ENV TARGET_ROCM_PATH=${ROCM_INSTALL_PATH} +ENV ROCM_VERSION=${ROCM_VERSION} +ENV GPU_FAMILY=${GPU_FAMILY} +ENV UBUNTU_VERSION=22.04 +ENV LD_LIBRARY_PATH=${ROCM_INSTALL_PATH}/lib/rocm_sysdeps/lib:${ROCM_INSTALL_PATH}/lib/llvm/lib:${ROCM_INSTALL_PATH}/lib + +WORKDIR /workspace diff --git a/.github/docker/rvs-nightly-rocm-ubuntu22.04/env-rocm.sh b/.github/docker/rvs-nightly-rocm-ubuntu22.04/env-rocm.sh new file mode 100755 index 000000000..d18dfb416 --- /dev/null +++ b/.github/docker/rvs-nightly-rocm-ubuntu22.04/env-rocm.sh @@ -0,0 +1,6 @@ +# Source in container shells: sets ROCm runtime paths (TheRock multiarch layout). +ROCM_PATH="${ROCM_PATH:-/opt/rocm/install}" +TARGET_ROCM_PATH="${TARGET_ROCM_PATH:-$ROCM_PATH}" +export ROCM_PATH TARGET_ROCM_PATH +export PATH="${ROCM_PATH}/bin:${PATH}" +export LD_LIBRARY_PATH="${ROCM_PATH}/lib/rocm_sysdeps/lib:${ROCM_PATH}/lib/llvm/lib:${ROCM_PATH}/lib:${LD_LIBRARY_PATH:-}" diff --git a/.github/docker/rvs-nightly-rocm-ubuntu22.04/setup-on-runner.sh b/.github/docker/rvs-nightly-rocm-ubuntu22.04/setup-on-runner.sh new file mode 100755 index 000000000..117b56f42 --- /dev/null +++ b/.github/docker/rvs-nightly-rocm-ubuntu22.04/setup-on-runner.sh @@ -0,0 +1,11 @@ +#!/usr/bin/env bash +# Run on the self-hosted GPU runner host to build the Ubuntu 22.04 ROCm-matched docker image. +# +# cd ROCmValidationSuite +# ./.github/docker/rvs-nightly-rocm-ubuntu22.04/setup-on-runner.sh --from-tarball amdrocm10-rvs-....tar.gz + +set -euo pipefail +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../../.." && pwd)" +cd "$ROOT" +chmod +x .github/docker/build-rocm-sdk-image.sh +exec .github/docker/build-rocm-sdk-image.sh --context .github/docker/rvs-nightly-rocm-ubuntu22.04 "$@" diff --git a/.github/docker/rvs-nightly-rocm-ubuntu26.04/Dockerfile b/.github/docker/rvs-nightly-rocm-ubuntu26.04/Dockerfile new file mode 100644 index 000000000..63ce8c714 --- /dev/null +++ b/.github/docker/rvs-nightly-rocm-ubuntu26.04/Dockerfile @@ -0,0 +1,43 @@ +# ROCm nightly-test runtime image on Ubuntu 26.04 — same SDK tarball layout as build_packages_local.sh +# (therock-dist-linux--.tar.gz from multiarch or release listing). +# +# Build on any docker host (typically the self-hosted runner): +# ./.github/docker/build-rocm-sdk-image.sh --context .github/docker/rvs-nightly-rocm-ubuntu26.04 \ +# --rocm-version 10.1.0a20260819 --gpu-family multiarch + +FROM ubuntu:26.04 + +ARG ROCM_VERSION +ARG GPU_FAMILY=multiarch +ARG ROCM_SDK_BASE_URL +ARG ROCM_INSTALL_PATH=/opt/rocm/install + +RUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends \ + ca-certificates \ + wget \ + tar \ + pciutils \ + kmod \ + python3 \ + && rm -rf /var/lib/apt/lists/* + +RUN test -n "${ROCM_VERSION}" && test -n "${ROCM_SDK_BASE_URL}" + +RUN mkdir -p "${ROCM_INSTALL_PATH}" \ + && wget -q -O /tmp/rocm-sdk.tar.gz \ + "${ROCM_SDK_BASE_URL%/}/therock-dist-linux-${GPU_FAMILY}-${ROCM_VERSION}.tar.gz" \ + && tar -xzf /tmp/rocm-sdk.tar.gz -C "${ROCM_INSTALL_PATH}" --strip-components=1 \ + && rm -f /tmp/rocm-sdk.tar.gz \ + && test -x "${ROCM_INSTALL_PATH}/bin/rocminfo" \ + && test -x "${ROCM_INSTALL_PATH}/bin/amd-smi" + +COPY env-rocm.sh /etc/profile.d/rocm-env.sh + +ENV ROCM_PATH=${ROCM_INSTALL_PATH} +ENV TARGET_ROCM_PATH=${ROCM_INSTALL_PATH} +ENV ROCM_VERSION=${ROCM_VERSION} +ENV GPU_FAMILY=${GPU_FAMILY} +ENV UBUNTU_VERSION=26.04 +ENV LD_LIBRARY_PATH=${ROCM_INSTALL_PATH}/lib/rocm_sysdeps/lib:${ROCM_INSTALL_PATH}/lib/llvm/lib:${ROCM_INSTALL_PATH}/lib + +WORKDIR /workspace diff --git a/.github/docker/rvs-nightly-rocm-ubuntu26.04/env-rocm.sh b/.github/docker/rvs-nightly-rocm-ubuntu26.04/env-rocm.sh new file mode 100644 index 000000000..d18dfb416 --- /dev/null +++ b/.github/docker/rvs-nightly-rocm-ubuntu26.04/env-rocm.sh @@ -0,0 +1,6 @@ +# Source in container shells: sets ROCm runtime paths (TheRock multiarch layout). +ROCM_PATH="${ROCM_PATH:-/opt/rocm/install}" +TARGET_ROCM_PATH="${TARGET_ROCM_PATH:-$ROCM_PATH}" +export ROCM_PATH TARGET_ROCM_PATH +export PATH="${ROCM_PATH}/bin:${PATH}" +export LD_LIBRARY_PATH="${ROCM_PATH}/lib/rocm_sysdeps/lib:${ROCM_PATH}/lib/llvm/lib:${ROCM_PATH}/lib:${LD_LIBRARY_PATH:-}" diff --git a/.github/docker/rvs-nightly-rocm-ubuntu26.04/setup-on-runner.sh b/.github/docker/rvs-nightly-rocm-ubuntu26.04/setup-on-runner.sh new file mode 100755 index 000000000..4b5694fc8 --- /dev/null +++ b/.github/docker/rvs-nightly-rocm-ubuntu26.04/setup-on-runner.sh @@ -0,0 +1,11 @@ +#!/usr/bin/env bash +# Run on the self-hosted GPU runner host to build the Ubuntu 26.04 ROCm-matched docker image. +# +# cd ROCmValidationSuite +# ./.github/docker/rvs-nightly-rocm-ubuntu26.04/setup-on-runner.sh --from-tarball amdrocm10-rvs-....tar.gz + +set -euo pipefail +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../../.." && pwd)" +cd "$ROOT" +chmod +x .github/docker/build-rocm-sdk-image.sh +exec .github/docker/build-rocm-sdk-image.sh --context .github/docker/rvs-nightly-rocm-ubuntu26.04 "$@" diff --git a/.github/docker/rvs-nightly-rocm/Dockerfile b/.github/docker/rvs-nightly-rocm/Dockerfile new file mode 100644 index 000000000..372860e2c --- /dev/null +++ b/.github/docker/rvs-nightly-rocm/Dockerfile @@ -0,0 +1,42 @@ +# ROCm nightly-test runtime image — same SDK tarball layout as build_packages_local.sh +# (therock-dist-linux--.tar.gz from multiarch or release listing). +# +# Build on any docker host (typically the self-hosted runner): +# ./.github/docker/build-rocm-sdk-image.sh --context .github/docker/rvs-nightly-rocm \ +# --rocm-version 10.1.0a20260819 --gpu-family multiarch + +FROM ubuntu:24.04 + +ARG ROCM_VERSION +ARG GPU_FAMILY=multiarch +ARG ROCM_SDK_BASE_URL +ARG ROCM_INSTALL_PATH=/opt/rocm/install + +RUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends \ + ca-certificates \ + wget \ + tar \ + pciutils \ + kmod \ + python3 \ + && rm -rf /var/lib/apt/lists/* + +RUN test -n "${ROCM_VERSION}" && test -n "${ROCM_SDK_BASE_URL}" + +RUN mkdir -p "${ROCM_INSTALL_PATH}" \ + && wget -q -O /tmp/rocm-sdk.tar.gz \ + "${ROCM_SDK_BASE_URL%/}/therock-dist-linux-${GPU_FAMILY}-${ROCM_VERSION}.tar.gz" \ + && tar -xzf /tmp/rocm-sdk.tar.gz -C "${ROCM_INSTALL_PATH}" --strip-components=1 \ + && rm -f /tmp/rocm-sdk.tar.gz \ + && test -x "${ROCM_INSTALL_PATH}/bin/rocminfo" \ + && test -x "${ROCM_INSTALL_PATH}/bin/amd-smi" + +COPY env-rocm.sh /etc/profile.d/rocm-env.sh + +ENV ROCM_PATH=${ROCM_INSTALL_PATH} +ENV TARGET_ROCM_PATH=${ROCM_INSTALL_PATH} +ENV ROCM_VERSION=${ROCM_VERSION} +ENV GPU_FAMILY=${GPU_FAMILY} +ENV LD_LIBRARY_PATH=${ROCM_INSTALL_PATH}/lib/rocm_sysdeps/lib:${ROCM_INSTALL_PATH}/lib/llvm/lib:${ROCM_INSTALL_PATH}/lib + +WORKDIR /workspace diff --git a/.github/docker/rvs-nightly-rocm/env-rocm.sh b/.github/docker/rvs-nightly-rocm/env-rocm.sh new file mode 100755 index 000000000..d18dfb416 --- /dev/null +++ b/.github/docker/rvs-nightly-rocm/env-rocm.sh @@ -0,0 +1,6 @@ +# Source in container shells: sets ROCm runtime paths (TheRock multiarch layout). +ROCM_PATH="${ROCM_PATH:-/opt/rocm/install}" +TARGET_ROCM_PATH="${TARGET_ROCM_PATH:-$ROCM_PATH}" +export ROCM_PATH TARGET_ROCM_PATH +export PATH="${ROCM_PATH}/bin:${PATH}" +export LD_LIBRARY_PATH="${ROCM_PATH}/lib/rocm_sysdeps/lib:${ROCM_PATH}/lib/llvm/lib:${ROCM_PATH}/lib:${LD_LIBRARY_PATH:-}" diff --git a/.github/docker/rvs-nightly-rocm/setup-on-runner.sh b/.github/docker/rvs-nightly-rocm/setup-on-runner.sh new file mode 100755 index 000000000..bb4009850 --- /dev/null +++ b/.github/docker/rvs-nightly-rocm/setup-on-runner.sh @@ -0,0 +1,12 @@ +#!/usr/bin/env bash +# Run on the self-hosted GPU runner host to build the ROCm-matched docker image. +# +# cd ROCmValidationSuite +# ./.github/docker/rvs-nightly-rocm/setup-on-runner.sh [--channel nightly|release] +# ./.github/docker/rvs-nightly-rocm/setup-on-runner.sh --from-tarball amdrocm7-rvs-....tar.gz + +set -euo pipefail +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../../.." && pwd)" +cd "$ROOT" +chmod +x .github/docker/build-rocm-sdk-image.sh +exec .github/docker/build-rocm-sdk-image.sh --context .github/docker/rvs-nightly-rocm "$@" diff --git a/.github/scripts/configure-rocm-sdk-channel.sh b/.github/scripts/configure-rocm-sdk-channel.sh new file mode 100644 index 000000000..6e6b17791 --- /dev/null +++ b/.github/scripts/configure-rocm-sdk-channel.sh @@ -0,0 +1,99 @@ +#!/usr/bin/env bash +# Configure ROCm SDK channel + ROCM_VERSION for build_packages_local.sh (CI). +# Uses POSIX case for release/* refs (safe under dash/sh and bash). +set -euo pipefail + +GITHUB_ENV="${GITHUB_ENV:-/dev/null}" + +NIGHTLY_BASE="${ROCM_SDK_NIGHTLY_BASE_URL:-}" +NIGHTLY_IDX="${ROCM_SDK_NIGHTLY_INDEX_URL:-}" +[ -z "$NIGHTLY_BASE" ] && NIGHTLY_BASE='https://nightly.repo.amd.com/rocm/core/tarball' +[ -z "$NIGHTLY_IDX" ] && NIGHTLY_IDX='https://nightly.repo.amd.com/rocm/core/tarball/' + +REL_URL="${ROCM_SDK_RELEASE_URL:-}" +[ -z "$REL_URL" ] && REL_URL='https://repo.amd.com/rocm/tarball/' +REL_BASE="${ROCM_SDK_RELEASE_BASE_URL:-}" +[ -z "$REL_BASE" ] && REL_BASE="${REL_URL%/}" + +EVENT="${GITHUB_EVENT_NAME:-}" +REF="${GITHUB_REF:-}" +IN_VER="${INPUT_ROCM_VERSION:-}" +VAR_VER="${VAR_ROCM_VERSION:-}" +IN_GPU="${INPUT_GPU_FAMILY:-}" + +release_build() { + { + echo "ROCM_SDK_CHANNEL=release" + echo "ROCM_SDK_RELEASE_URL=$REL_URL" + echo "ROCM_SDK_BASE_URL=$REL_BASE" + echo "ROCM_SDK_RELEASE_BASE_URL=$REL_BASE" + echo "ROCM_SDK_INDEX_URL=" + } >> "$GITHUB_ENV" +} + +nightly_build() { + { + echo "ROCM_SDK_CHANNEL=nightly" + echo "ROCM_SDK_RELEASE_URL=" + echo "ROCM_SDK_BASE_URL=$NIGHTLY_BASE" + echo "ROCM_SDK_INDEX_URL=$NIGHTLY_IDX" + } >> "$GITHUB_ENV" +} + +format_build() { + { + echo "ROCM_SDK_CHANNEL=auto" + echo "ROCM_SDK_RELEASE_URL=$REL_URL" + echo "ROCM_SDK_INDEX_URL=$NIGHTLY_IDX" + echo "ROCM_SDK_BASE_URL=$NIGHTLY_BASE" + echo "ROCM_SDK_NIGHTLY_BASE_URL=$NIGHTLY_BASE" + echo "ROCM_SDK_NIGHTLY_INDEX_URL=$NIGHTLY_IDX" + echo "ROCM_SDK_RELEASE_BASE_URL=$REL_BASE" + } >> "$GITHUB_ENV" +} + +# ROCM_VERSION: schedule clears; dispatch input; push/PR use repo variable when set. +if [ "$EVENT" = "schedule" ]; then + echo "ROCM_VERSION=" >> "$GITHUB_ENV" + echo "Scheduled run: build script will auto-fetch latest ROCm version" +elif [ -n "$IN_VER" ]; then + echo "ROCM_VERSION=$IN_VER" >> "$GITHUB_ENV" + echo "Using workflow_dispatch ROCm version: $IN_VER" +elif [ -n "$VAR_VER" ]; then + echo "ROCM_VERSION=$VAR_VER" >> "$GITHUB_ENV" + echo "Using repository variable ROCM_VERSION: $VAR_VER" +fi + +if [ -n "$IN_GPU" ]; then + echo "GPU_FAMILY=$IN_GPU" >> "$GITHUB_ENV" +fi + +if [ "$EVENT" = "schedule" ]; then + nightly_build + echo "ROCm SDK channel: nightly (scheduled — latest nightly)" +elif [ "$EVENT" = "pull_request" ]; then + release_build + echo "ROCm SDK channel: release (pull request — latest X.Y.Z)" +elif [ "$EVENT" = "push" ]; then + case "$REF" in + refs/heads/master|refs/heads/main|refs/heads/release/*) + release_build + echo "ROCm SDK channel: release (push to main or release/* — includes merge)" + ;; + *) + nightly_build + echo "ROCm SDK channel: nightly (push to feature branch)" + ;; + esac +elif [ "$EVENT" = "workflow_dispatch" ]; then + if [ -n "$IN_VER" ] || [ -n "$VAR_VER" ]; then + format_build + echo "ROCm SDK: manual — tarball base chosen by version format (input or vars.ROCM_VERSION)" + else + nightly_build + echo "ROCm SDK channel: nightly (manual, no version pin)" + fi +else + nightly_build + echo "ROCm SDK channel: nightly (fallback)" +fi diff --git a/.github/scripts/rvs-deb-unsigned-repo.sh b/.github/scripts/rvs-deb-unsigned-repo.sh new file mode 100644 index 000000000..2dfbb5e2a --- /dev/null +++ b/.github/scripts/rvs-deb-unsigned-repo.sh @@ -0,0 +1,168 @@ +#!/bin/sh +# Accumulate nightly/unsigned/packages/deb APT archive (dists/ + pool/) via reprepro. +# Merges this run's .deb into the existing S3 archive (no s3:DeleteObject / --delete). +# Usage: rvs-deb-unsigned-repo.sh [path-to-build-dir] +# Env: AWS_S3_BUCKET (required), optional RVS_UNSIGNED_DEB_PREFIX, GITHUB_RUN_ID +set -eu + +BUILD_DIR="${1:-./build}" +BUCKET="${AWS_S3_BUCKET:-}" +DEB_PREFIX="${RVS_UNSIGNED_DEB_PREFIX:-}" +META_OUT="${RVS_UNSIGNED_DEB_META_OUT:-${BUILD_DIR}/unsigned-deb-meta.json}" +RUN_ID="${GITHUB_RUN_ID:-}" + +SCRIPT_DIR=$(CDPATH= cd -- "$(dirname "$0")" && pwd) + +if [ -z "$BUCKET" ]; then + echo "::warning::AWS_S3_BUCKET not set. Skipping unsigned DEB repo update." + exit 0 +fi + +if [ -z "$DEB_PREFIX" ]; then + DEB_PREFIX=$(sh "${SCRIPT_DIR}/rvs-s3-upload-route.sh" unsigned-deb-prefix) +fi + +if ! command -v reprepro >/dev/null 2>&1; then + echo "Installing reprepro ..." + export DEBIAN_FRONTEND=noninteractive + apt-get update + apt-get install -y --no-install-recommends reprepro +fi + +DEBS=$(find "$BUILD_DIR" -maxdepth 1 -name 'amdrocm*-rvs*.deb' 2>/dev/null | sort) +if [ -z "$DEBS" ]; then + echo "::error::No amdrocm*-rvs*.deb in ${BUILD_DIR}; unsigned DEB publish requires a package." >&2 + exit 1 +fi + +STAGING=$(mktemp -d) +trap 'rm -rf "$STAGING"' EXIT + +mkdir -p "$STAGING/conf" +echo "Downloading existing unsigned DEB archive from s3://${BUCKET}/${DEB_PREFIX}/ ..." +aws s3 sync "s3://${BUCKET}/${DEB_PREFIX}/conf/" "$STAGING/conf/" --no-progress +aws s3 sync "s3://${BUCKET}/${DEB_PREFIX}/pool/" "$STAGING/pool/" --no-progress +aws s3 sync "s3://${BUCKET}/${DEB_PREFIX}/dists/" "$STAGING/dists/" --no-progress + +if [ ! -f "$STAGING/conf/distributions" ]; then + cat >"$STAGING/conf/distributions" <<'EOF' +Origin: ROCm Validation Suite +Label: stable +Suite: stable +Codename: stable +Architectures: amd64 +Components: main +Description: RVS unsigned nightly +EOF +fi + +# Track this run's pool keys for latest.json (not the entire historical pool). +NEW_META=$(mktemp) +trap 'rm -rf "$STAGING" "$NEW_META"' EXIT +: >"$NEW_META" + +deb_count=0 +for deb in $DEBS; do + deb_count=$((deb_count + 1)) + pkg=$(dpkg-deb -f "$deb" Package) + ver=$(dpkg-deb -f "$deb" Version) + # Same Package+Version already in the archive: drop that version from the local + # db/pool then re-include so PutObject can overwrite the pool object. + # Does not require s3:DeleteObject (orphan keys with other filenames may remain). + if reprepro -b "$STAGING" listfilter stable "Package (== ${pkg}), Version (== ${ver})" 2>/dev/null | grep -q .; then + echo "Package ${pkg} ${ver} already in suite stable; removing that version from local archive before re-include ..." + reprepro -b "$STAGING" -T deb removefilter stable "Package (== ${pkg}), Version (== ${ver})" + fi + echo "Including $(basename "$deb") ..." + reprepro -b "$STAGING" includedeb stable "$deb" + + # Resolve the pool path reprepro recorded for this exact Package+Version from + # its own Packages index. Using find is non-deterministic (undefined order) and + # a glob fallback on ${pkg}_*.deb would match every historical version in pool/. + PACKAGES_FILE="$STAGING/dists/stable/main/binary-amd64/Packages" + pool_rel=$(awk -v pkg="$pkg" -v ver="$ver" ' + /^Package:/ { cur_pkg=$2; cur_ver=""; cur_fn="" } + /^Version:/ { cur_ver=$2 } + /^Filename:/ { cur_fn=$2 } + /^$/ { if (cur_pkg==pkg && cur_ver==ver && cur_fn!="") { print cur_fn; cur_pkg=""; cur_ver=""; cur_fn="" } } + END { if (cur_pkg==pkg && cur_ver==ver && cur_fn!="") print cur_fn } + ' "$PACKAGES_FILE") + + if [ -z "$pool_rel" ]; then + echo "::error::Cannot resolve pool path for ${pkg} ${ver} from reprepro Packages index." >&2 + exit 1 + fi + echo "$pool_rel" >>"$NEW_META" +done + +if ! find "$STAGING/pool" -name '*.deb' 2>/dev/null | grep -q .; then + echo "::error::reprepro produced no packages under pool/; aborting before S3 sync." >&2 + exit 1 +fi + +meta_count=$(wc -l < "$NEW_META") +if [ "$meta_count" -ne "$deb_count" ]; then + echo "::error::Resolved ${meta_count} pool path(s) for ${deb_count} .deb file(s); counts must match." >&2 + exit 1 +fi + +echo "Uploading conf/, pool/, and dists/ to s3://${BUCKET}/${DEB_PREFIX}/ (accumulate; no --delete) ..." +aws s3 sync "$STAGING/conf/" "s3://${BUCKET}/${DEB_PREFIX}/conf/" --no-progress +aws s3 sync "$STAGING/pool/" "s3://${BUCKET}/${DEB_PREFIX}/pool/" --no-progress +aws s3 sync "$STAGING/dists/" "s3://${BUCKET}/${DEB_PREFIX}/dists/" --no-progress + +echo "=== Unsigned DEB archive updated at s3://${BUCKET}/${DEB_PREFIX}/ ===" +aws s3 ls "s3://${BUCKET}/${DEB_PREFIX}/dists/stable/" --human-readable 2>/dev/null || true + +# Meta for nightly/unsigned/latest.json: this run's packages only. +python3 <&2 + exit 1 +fi +aws s3 cp "$META_OUT" "s3://${BUCKET}/nightly/unsigned/runs/${RUN_ID}/deb.json" --no-progress diff --git a/.github/scripts/rvs-s3-upload-route.sh b/.github/scripts/rvs-s3-upload-route.sh new file mode 100644 index 000000000..04372ae6b --- /dev/null +++ b/.github/scripts/rvs-s3-upload-route.sh @@ -0,0 +1,259 @@ +#!/bin/sh +# POSIX S3 path routing for RVS packages (ubuntu:22.04 container uses sh/dash). +# Usage: +# rvs-s3-upload-route.sh upload-deb +# rvs-s3-upload-route.sh upload-rpm-tar +# rvs-s3-upload-route.sh deb-repo-prefix +# rvs-s3-upload-route.sh rpm-repo-prefix +# rvs-s3-upload-route.sh unsigned-deb-prefix +# rvs-s3-upload-route.sh unsigned-rpm-prefix +# rvs-s3-upload-route.sh unsigned-tar-prefix +# rvs-s3-upload-route.sh upload-rpm-tar-unsigned +# rvs-s3-upload-route.sh unsigned-upload-enabled +set -eu + +BASE="rvs" +EVENT="${GITHUB_EVENT_NAME:-}" +REF="${GITHUB_REF:-}" +REF_NAME="${GITHUB_REF_NAME:-}" +RUN_NUMBER="${GITHUB_RUN_NUMBER:-0}" +BUCKET="${AWS_S3_BUCKET:-}" +GITHUB_OUTPUT="${GITHUB_OUTPUT:-/dev/null}" + +RVS_SKIP_UPLOAD="${RVS_SKIP_UPLOAD:-false}" +RVS_IS_DEFAULT_BRANCH="${RVS_IS_DEFAULT_BRANCH:-false}" +RVS_BRANCH_PREFIX="${RVS_BRANCH_PREFIX:-}" +RVS_BUILD_REF_NAME="${RVS_BUILD_REF_NAME:-}" +RUNNER_SUFFIX="${RVS_RUNNER_SUFFIX:-ubuntu-22.04}" + +rvs_is_release_ref() { + case "$1" in + refs/heads/release/*) return 0 ;; + esac + return 1 +} + +rvs_unsigned_upload_enabled() { + [ "$EVENT" = "schedule" ] && [ "$RVS_IS_DEFAULT_BRANCH" = "true" ] +} + +rvs_resolve_route() { + RVS_S3_ROUTE="pr" + RVS_S3_DEB_PREFIX="" + RVS_S3_RPM_PREFIX="" + RVS_S3_TAR_PREFIX="" + RVS_S3_UNSIGNED_DEB_PREFIX="" + RVS_S3_UNSIGNED_RPM_PREFIX="" + RVS_S3_UNSIGNED_TAR_PREFIX="" + RVS_APT_SUITE="rvs-nightly" + RVS_S3_OUTPUT_PATHS="" + + if [ "$RVS_SKIP_UPLOAD" = "true" ]; then + RVS_S3_ROUTE="skip" + return 0 + fi + + if [ "$EVENT" = "schedule" ] && [ "$RVS_IS_DEFAULT_BRANCH" != "true" ]; then + RVS_S3_ROUTE="scheduled_branch" + RVS_S3_DEB_PREFIX="${RVS_BRANCH_PREFIX}/${RVS_BUILD_REF_NAME}/nightly/deb" + RVS_S3_RPM_PREFIX="${RVS_BRANCH_PREFIX}/${RVS_BUILD_REF_NAME}/nightly/rpm" + RVS_S3_TAR_PREFIX="${RVS_BRANCH_PREFIX}/${RVS_BUILD_REF_NAME}/nightly/tar" + RVS_S3_OUTPUT_PATHS="Ubuntu DEB|${RVS_BRANCH_PREFIX}/${RVS_BUILD_REF_NAME}/nightly/deb||CentOS/RHEL RPM|${RVS_BRANCH_PREFIX}/${RVS_BUILD_REF_NAME}/nightly/rpm||CentOS/RHEL TGZ|${RVS_BRANCH_PREFIX}/${RVS_BUILD_REF_NAME}/nightly/tar" + return 0 + fi + + if rvs_is_release_ref "$REF" && { [ "$EVENT" = "push" ] || [ "$EVENT" = "workflow_dispatch" ]; }; then + RVS_S3_ROUTE="release" + RVS_S3_DEB_PREFIX="release/${BASE}/deb" + RVS_S3_RPM_PREFIX="release/${BASE}/rpm" + RVS_S3_TAR_PREFIX="release/${BASE}/tar" + RVS_APT_SUITE="rvs-release" + RVS_S3_OUTPUT_PATHS="Ubuntu DEB|release/${BASE}/deb||CentOS/RHEL RPM|release/${BASE}/rpm||CentOS/RHEL TGZ|release/${BASE}/tar" + return 0 + fi + + if [ "$EVENT" = "schedule" ] || [ "$EVENT" = "push" ] || [ "$EVENT" = "workflow_dispatch" ]; then + RVS_S3_ROUTE="nightly" + RVS_S3_DEB_PREFIX="nightly/${BASE}/deb" + RVS_S3_RPM_PREFIX="nightly/${BASE}/rpm" + RVS_S3_TAR_PREFIX="nightly/${BASE}/tar" + RVS_APT_SUITE="rvs-nightly" + RVS_S3_OUTPUT_PATHS="Ubuntu DEB|nightly/${BASE}/deb||CentOS/RHEL RPM|nightly/${BASE}/rpm||CentOS/RHEL TGZ|nightly/${BASE}/tar" + if rvs_unsigned_upload_enabled; then + RVS_S3_UNSIGNED_DEB_PREFIX="nightly/unsigned/packages/deb" + RVS_S3_UNSIGNED_RPM_PREFIX="nightly/unsigned/packages/rpm/x86_64" + RVS_S3_UNSIGNED_TAR_PREFIX="nightly/unsigned/tarball" + RVS_S3_OUTPUT_PATHS="${RVS_S3_OUTPUT_PATHS}||Unsigned DEB|nightly/unsigned/packages/deb||Unsigned RPM|nightly/unsigned/packages/rpm/x86_64||Unsigned TGZ|nightly/unsigned/tarball" + fi + return 0 + fi + + RVS_S3_ROUTE="pr" + RVS_S3_DEB_PREFIX="${BASE}/${REF_NAME}/${RUN_NUMBER}/${RUNNER_SUFFIX}" + RVS_S3_RPM_PREFIX="${BASE}/${REF_NAME}/${RUN_NUMBER}/manylinux_2_28" + RVS_S3_TAR_PREFIX="${RVS_S3_RPM_PREFIX}" + RVS_S3_OUTPUT_PATHS="Ubuntu DEB|${RVS_S3_DEB_PREFIX}||CentOS/RHEL packages|${RVS_S3_RPM_PREFIX}" +} + +cmd="${1:-}" +rvs_resolve_route + +case "$cmd" in + upload-deb) + if [ -z "$BUCKET" ]; then + echo "::warning::AWS_S3_BUCKET not set. Skipping S3 upload." + exit 0 + fi + if [ "$RVS_S3_ROUTE" = "skip" ]; then + echo "Skipping S3 upload (scheduled release* branch)." + exit 0 + fi + case "$RVS_S3_ROUTE" in + scheduled_branch) + echo "Scheduled ACTIVE_BRANCHES build: uploading to ${RVS_S3_DEB_PREFIX}" + ;; + release) + echo "Release branch build: uploading to ${RVS_S3_DEB_PREFIX}" + ;; + nightly) + echo "Nightly/push build: uploading to ${RVS_S3_DEB_PREFIX}" + ;; + *) + echo "Uploading to s3://${BUCKET}/${RVS_S3_DEB_PREFIX}/" + ;; + esac + aws s3 cp ./build "s3://${BUCKET}/${RVS_S3_DEB_PREFIX}/" \ + --recursive --exclude "*" --include "amdrocm*-rvs*.deb" --no-progress + echo "Listing s3://${BUCKET}/${RVS_S3_DEB_PREFIX}/" + aws s3 ls "s3://${BUCKET}/${RVS_S3_DEB_PREFIX}/" --human-readable || true + echo "bucket=${BUCKET}" >> "$GITHUB_OUTPUT" + echo "paths=Ubuntu DEB|${RVS_S3_DEB_PREFIX}" >> "$GITHUB_OUTPUT" + echo "Done." + ;; + upload-rpm-tar) + if [ -z "$BUCKET" ]; then + echo "::warning::AWS_S3_BUCKET not set. Skipping S3 upload." + exit 0 + fi + if [ "$RVS_S3_ROUTE" = "skip" ]; then + echo "Skipping S3 upload (scheduled release* branch)." + exit 0 + fi + case "$RVS_S3_ROUTE" in + scheduled_branch) + echo "Scheduled ACTIVE_BRANCHES build: uploading to ${RVS_S3_RPM_PREFIX} and ${RVS_S3_TAR_PREFIX}" + ;; + release) + echo "Release branch build: uploading to ${RVS_S3_RPM_PREFIX} and ${RVS_S3_TAR_PREFIX}" + ;; + nightly) + echo "Nightly/push build: uploading to ${RVS_S3_RPM_PREFIX} and ${RVS_S3_TAR_PREFIX}" + ;; + *) + echo "Uploading to s3://${BUCKET}/${RVS_S3_RPM_PREFIX}/" + ;; + esac + if [ "$RVS_S3_ROUTE" = "pr" ]; then + aws s3 cp ./build "s3://${BUCKET}/${RVS_S3_RPM_PREFIX}/" \ + --recursive --exclude "*" --include "amdrocm*-rvs*.rpm" --include "amdrocm*-rvs*.tar.gz" --no-progress + echo "Listing s3://${BUCKET}/${RVS_S3_RPM_PREFIX}/" + aws s3 ls "s3://${BUCKET}/${RVS_S3_RPM_PREFIX}/" --human-readable || true + echo "bucket=${BUCKET}" >> "$GITHUB_OUTPUT" + echo "paths=CentOS/RHEL packages|${RVS_S3_RPM_PREFIX}" >> "$GITHUB_OUTPUT" + else + aws s3 cp ./build "s3://${BUCKET}/${RVS_S3_RPM_PREFIX}/" \ + --recursive --exclude "*" --include "amdrocm*-rvs*.rpm" --no-progress + aws s3 cp ./build "s3://${BUCKET}/${RVS_S3_TAR_PREFIX}/" \ + --recursive --exclude "*" --include "amdrocm*-rvs*.tar.gz" --no-progress + echo "Listing s3://${BUCKET}/${RVS_S3_RPM_PREFIX}/" + aws s3 ls "s3://${BUCKET}/${RVS_S3_RPM_PREFIX}/" --human-readable || true + echo "Listing s3://${BUCKET}/${RVS_S3_TAR_PREFIX}/" + aws s3 ls "s3://${BUCKET}/${RVS_S3_TAR_PREFIX}/" --human-readable || true + echo "bucket=${BUCKET}" >> "$GITHUB_OUTPUT" + echo "paths=CentOS/RHEL RPM|${RVS_S3_RPM_PREFIX}||CentOS/RHEL TGZ|${RVS_S3_TAR_PREFIX}" >> "$GITHUB_OUTPUT" + fi + echo "Done." + ;; + deb-repo-prefix) + echo "$RVS_S3_DEB_PREFIX" + echo "$RVS_APT_SUITE" + ;; + rpm-repo-prefix) + echo "$RVS_S3_RPM_PREFIX" + ;; + unsigned-upload-enabled) + if rvs_unsigned_upload_enabled; then + echo "true" + else + echo "false" + fi + ;; + unsigned-deb-prefix) + if [ -z "$RVS_S3_UNSIGNED_DEB_PREFIX" ]; then + echo "::error::Unsigned DEB upload is only enabled for scheduled default-branch builds." >&2 + exit 1 + fi + echo "$RVS_S3_UNSIGNED_DEB_PREFIX" + ;; + unsigned-rpm-prefix) + if [ -z "$RVS_S3_UNSIGNED_RPM_PREFIX" ]; then + echo "::error::Unsigned RPM upload is only enabled for scheduled default-branch builds." >&2 + exit 1 + fi + echo "$RVS_S3_UNSIGNED_RPM_PREFIX" + ;; + unsigned-tar-prefix) + if [ -z "$RVS_S3_UNSIGNED_TAR_PREFIX" ]; then + echo "::error::Unsigned TAR upload is only enabled for scheduled default-branch builds." >&2 + exit 1 + fi + echo "$RVS_S3_UNSIGNED_TAR_PREFIX" + ;; + upload-rpm-tar-unsigned) + if [ -z "$BUCKET" ]; then + echo "::warning::AWS_S3_BUCKET not set. Skipping unsigned S3 upload." + exit 0 + fi + if [ -z "$RVS_S3_UNSIGNED_RPM_PREFIX" ] || [ -z "$RVS_S3_UNSIGNED_TAR_PREFIX" ]; then + echo "Skipping unsigned S3 upload (not a scheduled default-branch build)." + exit 0 + fi + rpm_count=0 + for f in ./build/amdrocm*-rvs*.rpm; do + [ -f "$f" ] || continue + rpm_count=$((rpm_count + 1)) + done + tar_count=0 + for f in ./build/amdrocm*-rvs*.tar.gz; do + [ -f "$f" ] || continue + tar_count=$((tar_count + 1)) + if [ ! -f "${f}.sha256" ]; then + echo "::error::Missing SHA-256 sidecar for $(basename "$f"); run sha256sum before unsigned upload." >&2 + exit 1 + fi + done + if [ "$rpm_count" -lt 1 ]; then + echo "::error::No amdrocm*-rvs*.rpm in ./build; refusing unsigned RPM upload." >&2 + exit 1 + fi + if [ "$tar_count" -lt 1 ]; then + echo "::error::No amdrocm*-rvs*.tar.gz in ./build; refusing unsigned TAR upload." >&2 + exit 1 + fi + + echo "Scheduled unsigned build: accumulating into ${RVS_S3_UNSIGNED_RPM_PREFIX} and ${RVS_S3_UNSIGNED_TAR_PREFIX}" + aws s3 cp ./build "s3://${BUCKET}/${RVS_S3_UNSIGNED_RPM_PREFIX}/" \ + --recursive --exclude "*" --include "amdrocm*-rvs*.rpm" --no-progress + aws s3 cp ./build "s3://${BUCKET}/${RVS_S3_UNSIGNED_TAR_PREFIX}/" \ + --recursive --exclude "*" --include "amdrocm*-rvs*.tar.gz" --include "amdrocm*-rvs*.tar.gz.sha256" --no-progress + echo "Listing s3://${BUCKET}/${RVS_S3_UNSIGNED_RPM_PREFIX}/" + aws s3 ls "s3://${BUCKET}/${RVS_S3_UNSIGNED_RPM_PREFIX}/" --human-readable || true + echo "Listing s3://${BUCKET}/${RVS_S3_UNSIGNED_TAR_PREFIX}/" + aws s3 ls "s3://${BUCKET}/${RVS_S3_UNSIGNED_TAR_PREFIX}/" --human-readable || true + echo "Done." + ;; + *) + echo "Usage: $0 upload-deb|upload-rpm-tar|deb-repo-prefix|rpm-repo-prefix|unsigned-deb-prefix|unsigned-rpm-prefix|unsigned-tar-prefix|upload-rpm-tar-unsigned|unsigned-upload-enabled" >&2 + exit 1 + ;; +esac diff --git a/.github/scripts/rvs-unsigned-publish-latest.sh b/.github/scripts/rvs-unsigned-publish-latest.sh new file mode 100644 index 000000000..ca1dc5b6c --- /dev/null +++ b/.github/scripts/rvs-unsigned-publish-latest.sh @@ -0,0 +1,79 @@ +#!/bin/sh +# Merge per-run unsigned metadata and publish nightly/unsigned/latest.json for signing CI. +# Env: AWS_S3_BUCKET (required), GITHUB_RUN_ID, GITHUB_SHA, ROCM_VERSION (optional) +set -eu + +BUCKET="${AWS_S3_BUCKET:-}" +RUN_ID="${GITHUB_RUN_ID:-}" +SHA="${GITHUB_SHA:-}" +ROCM_VERSION="${ROCM_VERSION:-}" + +if [ -z "$BUCKET" ]; then + echo "::warning::AWS_S3_BUCKET unset; skipping latest.json publish." + exit 0 +fi + +if [ -z "$RUN_ID" ]; then + echo "::error::GITHUB_RUN_ID unset; cannot publish latest.json." >&2 + exit 1 +fi + +WORKDIR=$(mktemp -d) +trap 'rm -rf "$WORKDIR"' EXIT + +DEB_JSON="$WORKDIR/deb.json" +RPM_JSON="$WORKDIR/rpm-tar.json" + +if ! aws s3 cp "s3://${BUCKET}/nightly/unsigned/runs/${RUN_ID}/deb.json" "$DEB_JSON" --no-progress; then + echo "::error::Missing s3://${BUCKET}/nightly/unsigned/runs/${RUN_ID}/deb.json (unsigned DEB step)." >&2 + exit 1 +fi + +if ! aws s3 cp "s3://${BUCKET}/nightly/unsigned/runs/${RUN_ID}/rpm-tar.json" "$RPM_JSON" --no-progress; then + echo "::error::Missing s3://${BUCKET}/nightly/unsigned/runs/${RUN_ID}/rpm-tar.json (unsigned RPM/TAR step)." >&2 + exit 1 +fi + +python3 <.date, PRs add .branch.commit) + │ ├── export CPACK_RPM_PACKAGE_RELEASE (same as DEB) + │ ├── export CMAKE_CXX_COMPILER=hipcc (AlmaLinux and Ubuntu/Debian) + │ └── export CMAKE_COMMAND=cmake3 (AlmaLinux) or cmake (Ubuntu) + ├── 4. Configure CMake (install RPATH defaults live in CMakeLists.txt) + │ └── $CMAKE_COMMAND -B ./build \ + │ -DCMAKE_BUILD_TYPE=Release \ + │ -DROCM_PATH=$ROCM_PATH \ + │ -DHIP_PLATFORM=amd \ + │ -DCMAKE_CXX_COMPILER=$CMAKE_CXX_COMPILER (if set) \ + │ -DROCM_MAJOR_VERSION=$ROCM_MAJOR \ + │ -DCMAKE_INSTALL_PREFIX=/opt/rocm/extras-$ROCM_MAJOR \ + │ -DCPACK_PACKAGING_INSTALL_PREFIX=/opt/rocm/extras-$ROCM_MAJOR \ + │ -DCMAKE_VERBOSE_MAKEFILE=1 \ + │ -DFETCH_ROCMPATH_FROM_ROCMCORE=ON + ├── 5. Build RVS + │ └── make -C ./build -j$(nproc) + └── 6. Create Packages + ├── Ubuntu: DEB + TGZ (via CPack) + └── AlmaLinux: RPM + TGZ (via CPack) +``` + +### Key Technical Details + +**Relocatable RPATH**: The canonical list lives in **[`cmake_modules/RVSPackagedRpath.cmake`](../../cmake_modules/RVSPackagedRpath.cmake)** (host triple detection via `amdclang++ --print-target-triple` is done there) and is applied via **`CMAKE_INSTALL_RPATH`** / **`CMAKE_BUILD_RPATH`** in **`CMakeLists.txt`**. **`CMAKE_*_LINKER_FLAGS_INIT`** only adds **`--enable-new-dtags`** (RUNPATH behavior). **Local dev builds** may also retain implicit **`$ROCM_PATH`** link-dir RUNPATH entries ( **`CMAKE_INSTALL_REMOVE_ENVIRONMENT_RPATH`** is set only when **`GITHUB_ACTIONS=true`** ). + +**Packaged artifacts** (DEB/RPM/TGZ): **`CPACK_PRE_BUILD_SCRIPTS`** runs **[`cmake_modules/cpack-patch-rpath.cmake.in`](../../cmake_modules/cpack-patch-rpath.cmake.in)** before CPack seals the package. It uses **`patchelf`** to replace RUNPATH on every staged ELF with the canonical list (no build-machine SDK paths). Requires **`patchelf`** on the packaging host (`build_packages_local.sh` installs it). + +The **`$ORIGIN`** relative entries resolve to the install prefix (`.../extras-/bin` → `.../extras-/lib`). Absolute paths add **`/opt/rocm/lib`**, **`/opt/rocm/lib/llvm/lib`**, **`/opt/rocm/core-/lib`**, **`/opt/rocm/core-/lib/llvm/lib`**, and per-host-triple **`libomp`** dirs when the triple is known — equivalent to: + +```bash +# Example for ROCm major 10 (per-triple libomp dirs omitted) +CMAKE_INSTALL_RPATH='$ORIGIN:$ORIGIN/../lib:$ORIGIN/../lib/rvs:/opt/rocm/lib:/opt/rocm/lib/llvm/lib:/opt/rocm/core-10/lib:/opt/rocm/core-10/lib/llvm/lib' +``` + +**Automatic Version Management**: CMake reads the project version from `CMakeLists.txt` and CPack uses it for package naming automatically. The **patch version** is auto-computed at CMake configure time: `git describe --tags --match "v..*"` counts commits since the last matching `v` tag. For example, if the tag is `v1.3.0` and there have been 15 commits since, the package version becomes `1.3.15`. If no matching tag exists or git is unavailable, the patch defaults to `0` from `CMakeLists.txt`. This works for both CI builds and direct local `cmake` invocations. + +**HIP Device Libraries**: Probes known TheRock layouts (`lib/llvm/amdgcn/bitcode`, legacy `amdgcn/bitcode`, Clang resource-dir `lib/llvm/lib/clang/*/lib/amdgcn/bitcode`) and falls back to `amdclang++ -print-resource-dir`; exports `HIP_DEVICE_LIB_PATH` for clang device library discovery. + +**Compiler Selection**: `CMAKE_CXX_COMPILER` is set to ROCm's `hipcc` on both AlmaLinux (manylinux_2_28) and Ubuntu/Debian. Because hipcc is Clang-based and does not bundle libstdc++, the system GCC tree is plumbed in via `--gcc-toolchain=$GCC_TOOLCHAIN`: on AlmaLinux this points at the discovered `gcc-toolset-`; on Ubuntu/Debian it is `/usr` (set when `/usr/include/c++/*/barrier` exists). + +**CMake Command**: Uses `cmake3` on AlmaLinux (manylinux_2_28) and `cmake` on Ubuntu. + +**Verbose Build Output**: `CMAKE_VERBOSE_MAKEFILE=1` enables detailed compilation output for debugging and transparency. + +**Dynamic ROCm Path Discovery**: `FETCH_ROCMPATH_FROM_ROCMCORE=ON` allows RVS to automatically detect ROCm installation location at runtime from ROCm core libraries. + +**Single Source of Truth**: `build_packages_local.sh` drives configure/build/package for CI and local use; **RPATH defaults** live in **`CMakeLists.txt`** so a plain **`cmake`** invocation gets the same install **`RPATH`** without copying flags into the shell script. When **`BUILD_TRANSFERBENCH_CLI=ON`**, [`CMakeTransferBenchCLI.cmake`](../CMakeTransferBenchCLI.cmake) applies TransferBench-specific relocatable RPATH via [`CMakeTransferBenchRPATH.cmake.in`](../CMakeTransferBenchRPATH.cmake.in), forwards **`GPU_TARGETS`** (via `-C` initial cache) / **`HIP_PLATFORM=amd`**, and prints **`message(STATUS)`** lines for the sub-build. **CI** streams verbose TransferBench `cmake --build --verbose` output in the job log (`LOG_BUILD=OFF`); local builds keep stamp logs under `build/TransferBenchCLI-prefix/src/TransferBenchCLI-stamp/`. + +### Workflow Steps + +The GitHub Actions workflow performs minimal platform-specific operations: + +1. **Checkout Repository** with recursive submodule initialization +2. **Set Environment Variables** from workflow inputs or defaults +3. **Execute Build Script** - `./build_packages_local.sh` handles everything +4. **Verify Packages** - Platform-specific verification (dpkg-deb or rpm -q) +5. **Upload to S3** (when the repo is `ROCm/ROCmValidationSuite`, or when repository variable `RVS_S3_UPLOAD_ENABLED` is `true`) – Each job uploads its packages to S3 using OIDC. The bash routing logic determines the S3 path: `release/*` branch builds (push or manual) go to `release/`, scheduled/push/manual builds go to `nightly/`, and PR builds go to a ref-specific path. Requires `AWS_S3_BUCKET` (variable) and `AWS_ROLE_ARN` (secret). Skipped gracefully if `AWS_S3_BUCKET` is not set. +6. **Generate Repo Metadata** (schedule, push, and manual builds only) – Creates APT repo metadata (`Packages`, `Packages.gz`, `Release`) for DEB and YUM/DNF repodata (`repodata/`) for RPM under `nightly/rvs/` (or `release/rvs/`), then uploads to S3 so the paths can be used as native package repositories. Skipped for PR builds since their packages go to one-off ref-specific paths. +7. **Unsigned nightly publish** (**scheduled default branch only**) – Accumulates into `nightly/unsigned/packages/deb/` (`dists/` + `pool/`, suite **`stable main`**) via [rvs-deb-unsigned-repo.sh](../scripts/rvs-deb-unsigned-repo.sh), `nightly/unsigned/packages/rpm/x86_64/` (`createrepo_c`, merge existing RPMs), and `nightly/unsigned/tarball/` (`.tar.gz` plus `.tar.gz.sha256` sidecars), then **`publish-unsigned-latest`** writes `nightly/unsigned/latest.json`. Phase 1 dual-write with `nightly/rvs/*` continues for legacy consumers. + +### S3 Upload (OIDC – No Stored Credentials) + +S3 upload runs when the repository is **`ROCm/ROCmValidationSuite`** **or** when **`RVS_S3_UPLOAD_ENABLED`** is set to the literal string **`true`** (for forks, mirrors, or other org repositories that opt in after IAM trust is updated). Fork PRs from other repositories are still skipped (no OIDC for cross-repo PR heads). The upload step is reached when those guards pass, but exits gracefully if `AWS_S3_BUCKET` is not set. Uses **AWS OIDC**; no long-term access key or secret. The bash routing inside the upload step determines the S3 destination based on the event type and branch. + +**Where `vars` and `secrets` are defined** + +They are **not** in the workflow file. They are set in the repo: + +- **Secrets** (e.g. `secrets.AWS_ROLE_ARN`): **Settings** → **Secrets and variables** → **Actions** → **Secrets** tab → New repository secret. +- **Variables** (e.g. `vars.AWS_S3_BUCKET`): **Settings** → **Secrets and variables** → **Actions** → **Variables** tab → New repository variable. + +### Runner Configuration + +The workflow uses GitHub repository variables to control which runners execute each job, allowing you to use self-hosted runners or GitHub-provided runners: + +| Variable | Default | Used by | +|----------|---------|---------| +| `ACTIVE_BRANCHES` | _(unset)_ | **Scheduled** nightly only: comma-separated branch literals/globs (`npi/**`, `release/**`). Default branch is always built; do not list it here. `release*` matches build but do not upload on schedule. | +| `RUNNER_LABEL` | `ubuntu-22.04` | Ubuntu build job | +| `RUNNER_LABEL_CONTAINER` | `ubuntu-latest` | CentOS/manylinux build job (container) | +| `RUNNER_LABEL_UTILITY` | `ubuntu-latest` | Release summary job | +| `RVS_S3_UPLOAD_ENABLED` | _(unset)_ | Set to `true` to run S3/OIDC upload steps when the repo is not `ROCm/ROCmValidationSuite`. Requires `AWS_S3_BUCKET`, `AWS_ROLE_ARN`, and IAM trust for this repository. | + +To use a self-hosted runner, set the variable to your runner's label (e.g., `self-hosted` or a custom label) in **Settings** → **Secrets and variables** → **Actions** → **Variables**. + +**Required setup for S3 upload:** + +1. **Repository secret** (Secrets tab, see above): + - Name: `AWS_ROLE_ARN` + - Value: the IAM role ARN to assume for S3 upload (e.g. `arn:aws:iam::123456789012:role/my-s3-upload-role`). Keeps the role ID hidden from the workflow. + +2. **Repository variable** (Variables tab, see above): + - Name: `AWS_S3_BUCKET` + - Value: your S3 bucket name (e.g. `my-rocm-packages`). + +3. **Repository variable** (only if the repo is **not** `ROCm/ROCmValidationSuite` and you want S3 upload from this repo): + - Name: `RVS_S3_UPLOAD_ENABLED` + - Value: `true` (must be this exact string). Upstream does not need this variable. + +4. **AWS IAM**: The role in `AWS_ROLE_ARN` must have a trust policy allowing GitHub OIDC to assume it for the **repository that runs the workflow** (identity provider `token.actions.githubusercontent.com`, audience `sts.amazonaws.com`) and permissions to `s3:PutObject`, `s3:GetObject`, and `s3:ListBucket` on the bucket, including prefix **`nightly/unsigned/`**. Unsigned publish **accumulates** objects and does **not** require `s3:DeleteObject`. + +**S3 path layout** (resolved by [`.github/scripts/rvs-s3-upload-route.sh`](../scripts/rvs-s3-upload-route.sh), POSIX-safe for Ubuntu `sh`): + +| Trigger | Path | Contents | +|--------|------|----------| +| **`release/*` branch** (`push` or `workflow_dispatch`) | `release/rvs/deb/`, `release/rvs/rpm/`, `release/rvs/tar/` | DEB → `.../deb` (Ubuntu job); RPM and TGZ → `.../rpm` and `.../tar` (manylinux job). Only PR merges into release branches or manual dispatch on release branches write here. | +| **Scheduled** (default branch only) | `nightly/rvs/deb/`, `nightly/rvs/rpm/`, `nightly/rvs/tar/` | Flat APT/YUM metadata (legacy consumer paths). **Also** `nightly/unsigned/packages/deb/`, `nightly/unsigned/packages/rpm/x86_64/`, `nightly/unsigned/tarball/` for signing CI (see below). | +| **Scheduled** (`ACTIVE_BRANCHES`, non-default, not `release*`) | `{branch_prefix}/{branch}/nightly/deb/`, `…/rpm/`, `…/tar/` | No shared `rvs/` segment; no repo metadata on these paths. | +| **Scheduled** (`release*` from `ACTIVE_BRANCHES`) | _(none)_ | Build only; upload skipped. | +| **Push to `master`/`main`**, or **`workflow_dispatch` on non-release branch** | `nightly/rvs/deb/`, `nightly/rvs/rpm/`, `nightly/rvs/tar/` | Same split by type. | +| **Pull request** (same-repo) | `rvs///ubuntu-22.04/` or `.../manylinux_2_28/` | DEB only (Ubuntu job); RPM+TGZ (manylinux job). One-off path, no repo metadata generated. | + +If `AWS_S3_BUCKET` is not set, the upload step is skipped with a warning (the workflow still succeeds). + +When packages are uploaded to S3, the **build report** artifact includes an **S3 Upload Locations** section with clickable links to each S3 prefix (AWS Console). This makes it easy to open the bucket and browse the uploaded DEB, RPM, and TGZ files from the report. + +### Unsigned nightly (`nightly/unsigned/`) — scheduled default branch + +Separate signing CI consumes **unsigned** packages from this prefix. Each scheduled default-branch run **accumulates** new `.deb`/`.rpm`/`.tar.gz` under `nightly/unsigned/` and regenerates APT/YUM metadata (merge existing objects + this run’s packages). Historical packages remain in the prefix (the OIDC role does **not** use `s3:DeleteObject`). A **`nightly/unsigned/latest.json`** pointer is published after both package jobs succeed; signing CI should read that file for exact `s3_key` / `sha256` values for **this run only**. Per-run fragments live under `nightly/unsigned/runs//deb.json` and `rpm-tar.json`. + +**Strict contract (scheduled default branch):** Unsigned steps **fail the workflow** if a required `.deb`, `.rpm`, or `.tar.gz` (and `.tar.gz.sha256` sidecar) is missing, if run metadata fragments are missing, or if `latest.json` validation fails. Exit 0 without publishing only when S3 upload is disabled (`AWS_S3_BUCKET` unset) or unsigned routing does not apply. A green **`publish-unsigned-latest`** job means `latest.json` points at this run’s objects under the accumulated `nightly/unsigned/` tree. + +**Rollout:** Phase 1 (current) **dual-writes** on schedule: legacy `nightly/rvs/*` plus `nightly/unsigned/*`. Phase 2 (future): scheduled builds write only `nightly/unsigned/*`; signed packages are published to consumer paths by signing CI. + +**S3 layout** (AMD-style DEB archive, same shape as [stable.repo.amd.com/rocm/core/packages/ubuntu2204/](https://stable.repo.amd.com/rocm/core/packages/ubuntu2204/) but under a single `deb/` prefix): + +``` +s3:///nightly/unsigned/ +├── packages/ +│ ├── deb/ +│ │ ├── conf/ # reprepro state (internal; not for apt clients) +│ │ ├── pool/main/…/amdrocm*-rvs_*.deb +│ │ └── dists/stable/ +│ │ ├── Release +│ │ └── main/binary-amd64/Packages(.gz) +│ └── rpm/ +│ └── x86_64/ +│ ├── amdrocm*-rvs*.rpm +│ └── repodata/ +├── tarball/ +│ ├── amdrocm*-rvs*-Linux.tar.gz +│ └── amdrocm*-rvs*-Linux.tar.gz.sha256 +├── latest.json # signing CI: exact keys + digests for this nightly run +└── runs// + ├── deb.json + └── rpm-tar.json +``` + +**Scripts:** [rvs-s3-upload-route.sh](../scripts/rvs-s3-upload-route.sh) (`upload-rpm-tar-unsigned` validates RPM+TGZ+sidecar then `aws s3 cp`); [rvs-deb-unsigned-repo.sh](../scripts/rvs-deb-unsigned-repo.sh) (merge into existing `reprepro` archive, sync without `--delete`); [rvs-unsigned-publish-latest.sh](../scripts/rvs-unsigned-publish-latest.sh) (merge run metadata → `latest.json`, fail on missing/invalid input). + +**apt (unsigned staging, internal testing):** + +```bash +echo "deb [trusted=yes arch=amd64] https://.s3.amazonaws.com/nightly/unsigned/packages/deb/ stable main" \ + | sudo tee /etc/apt/sources.list.d/rvs-unsigned-nightly.list +sudo apt update +``` + +**yum/dnf (unsigned RPM):** + +```bash +cat <<'EOF' | sudo tee /etc/yum.repos.d/rvs-unsigned-nightly.repo +[rvs-unsigned-nightly] +name=RVS Unsigned Nightly RPM +baseurl=https://.s3.amazonaws.com/nightly/unsigned/packages/rpm/x86_64/ +enabled=1 +gpgcheck=0 +EOF +``` + +**Tarball integrity:** Tarballs are not signed by signing CI. Each `.tar.gz` under `nightly/unsigned/tarball/` has a GNU **`sha256sum`** sidecar (`.tar.gz.sha256`). After download: + +```bash +cd /path/to/download +sha256sum -c amdrocm*-rvs*.tar.gz.sha256 +``` + +Checksums detect corruption or wrong files; they do not authenticate the publisher (use HTTPS and bucket IAM for that). + +**Signing CI handoff (out of this repo):** Read **`s3:///nightly/unsigned/latest.json`** (updated by the `publish-unsigned-latest` job after a successful scheduled build). It lists `deb.packages[].s3_key`, `rpm.s3_key`, and SHA-256 digests. Trigger via `workflow_run`, S3 event on `latest.json`, or manual dispatch with `github_run_id`. Signed `.deb`/`.rpm` are promoted to consumer repos (for example `nightly/rvs/` or AMD CDN) with appropriate signed metadata. + +**Unsigned DEB accumulate semantics:** Each run syncs existing `conf/` + `pool/` + `dists/` from S3, runs `reprepro includedeb` for this night’s `.deb` (if the same Package+Version is already present locally, it is removed from the **local** archive first so the pool object can be overwritten via PutObject), then syncs back **without** `--delete`. Older differently versioned packages remain in the bucket. Signing CI must use **`latest.json`**, not “newest object in the prefix.” + +### Repository Metadata (repodata) + +For **scheduled**, **push**, and **manual** (`workflow_dispatch`) builds, the workflow generates package repository metadata so that the S3 paths can be used directly as `apt` (DEB) and `yum`/`dnf` (RPM) repositories. This runs after the package upload step in each job. PR builds are excluded since their packages go to one-off ref-specific paths. + +**RPM repodata** (CentOS/RHEL job): +- Tool: `createrepo_c --simple-md-filenames --no-database --compress-type gz` (falls back to `createrepo`, which already uses short gzip names) +- Downloads existing RPMs from S3, merges in the newly built RPM, regenerates the `repodata/` directory, and syncs everything back +- Result matches [stable extras repodata](https://stable.repo.amd.com/rocm/extras/rvs/packages/rhel8/x86_64/repodata/) except the signature: `repodata/repomd.xml`, `repodata/primary.xml.gz`, `repodata/filelists.xml.gz`, `repodata/other.xml.gz`. No checksum-prefixed names, sqlite, or zstd. `repomd.xml.asc` is added later by the signing job. + +**DEB repo metadata** (Ubuntu job): +- Tools: `dpkg-scanpackages`, `apt-ftparchive` +- Downloads existing DEBs from S3, merges in the newly built DEB, regenerates `Packages`, `Packages.gz`, and `Release`, and syncs everything back +- Result: `Packages`, `Packages.gz`, `Release` + +**S3 directory layout after metadata generation:** + +``` +s3:///nightly/rvs/ +├── deb/ +│ ├── amdrocm7-rvs_1.3.15-r0711.20260423_amd64.deb +│ ├── Packages +│ ├── Packages.gz +│ └── Release +├── rpm/ +│ ├── amdrocm7-rvs-1.3.15-r0711.20260423.x86_64.rpm +│ └── repodata/ +│ ├── repomd.xml +│ ├── primary.xml.gz +│ ├── filelists.xml.gz +│ └── other.xml.gz +└── tar/ + └── amdrocm7-rvs-1.3.15-r0711.20260423-Linux.tar.gz +``` + +**Using the S3 repo with apt (Ubuntu/Debian):** + +```bash +# Add the nightly repo (replace with the actual S3 bucket name). +# Use "/" as the suite field for this flat repo layout. +echo "deb [trusted=yes] https://.s3.amazonaws.com/nightly/rvs/deb /" \ + | sudo tee /etc/apt/sources.list.d/rvs-nightly.list + +# Or the release repo +echo "deb [trusted=yes] https://.s3.amazonaws.com/release/rvs/deb /" \ + | sudo tee /etc/apt/sources.list.d/rvs-release.list + +sudo apt update +sudo apt install amdrocm7-rvs # Replace 7 with your ROCm major version +``` + +After `apt update`, **`apt list`** should show **`rvs-nightly`** or **`rvs-release`** (matching the bucket: `nightly/rvs/deb` vs `release/rvs/deb`) in the suite/codename column, because CI sets `Suite`, `Label`, and `Codename` in the flat `Release` file via `apt-ftparchive` options. If you still see **`unknown`**, run `apt update` again after a fresh metadata upload, or check that your `Release` on the server includes those fields. **`apt install amdrocm7-rvs`** still resolves versions from `Packages` either way. + +**Using the S3 repo with yum/dnf (CentOS/RHEL/Rocky):** + +```bash +# Add the nightly repo (replace with the actual S3 bucket name) +cat <<'EOF' | sudo tee /etc/yum.repos.d/rvs-nightly.repo +[rvs-nightly] +name=RVS Nightly Packages +baseurl=https://.s3.amazonaws.com/nightly/rvs/rpm/ +enabled=1 +gpgcheck=0 +EOF + +# Or the release repo +cat <<'EOF' | sudo tee /etc/yum.repos.d/rvs-release.repo +[rvs-release] +name=RVS Release Packages +baseurl=https://.s3.amazonaws.com/release/rvs/rpm/ +enabled=1 +gpgcheck=0 +EOF + +sudo yum install amdrocm7-rvs # Replace 7 with your ROCm major version +# or: sudo dnf install amdrocm7-rvs +``` + +> **Note:** `[trusted=yes]` (apt) and `gpgcheck=0` (yum) disable GPG verification. For production use, sign the packages and metadata with a GPG key and distribute the public key to users. Repository metadata is generated for **scheduled**, **push**, and **manual** builds; PR builds upload raw packages to ref-specific paths without metadata. + +## Build Script: build_packages_local.sh + +The workflow uses `build_packages_local.sh` as the core build engine. This script provides a complete, self-contained build system that works identically in both local development and CI/CD environments. + +### Script Features + +- **OS Detection**: Automatically identifies Ubuntu vs AlmaLinux +- **Dependency Management**: Installs all required build tools and libraries + - Enables PowerTools/CRB repository on AlmaLinux for doxygen and yaml-cpp +- **ROCm SDK Setup**: Auto-fetches latest version or downloads specified version from TheRock tarballs +- **HIP Device Libraries**: Ordered candidate-path probe (no full-tree `find`); exports `HIP_DEVICE_LIB_PATH` +- **CMake Configuration**: Sets up relocatable RPATHs and all build parameters + - Uses cmake3 on AlmaLinux, cmake on Ubuntu + - Sets CMAKE_CXX_COMPILER to hipcc on AlmaLinux and Ubuntu/Debian (Ubuntu also gets `--gcc-toolchain=/usr` for C++20 libstdc++ headers) +- **Building**: Compiles RVS with parallel builds +- **Packaging**: Creates DEB, RPM, and TGZ packages using CPack +- **Color Output**: Clear, colored progress indicators +- **Error Handling**: Robust error checking at each step + +### Running Locally + +```bash +# Basic usage (automatically installs dependencies, fetches latest ROCm) +sudo ./build_packages_local.sh + +# Custom ROCm version and GPU family (use sudo -E to preserve environment variables) +sudo -E ROCM_VERSION=7.11.0a20260121 GPU_FAMILY=gfx110X-all ./build_packages_local.sh + +# Or export first, then run with sudo -E +export ROCM_VERSION=7.11.0a20260121 +export GPU_FAMILY=gfx110X-all +sudo -E ./build_packages_local.sh + +# Debug build +sudo BUILD_TYPE=Debug ./build_packages_local.sh +``` + +**Important**: The script requires root privileges to install system dependencies. Use `sudo` when running locally on a bare-metal host. In GitHub Actions every build job runs inside a container as root, so the workflow invokes `./build_packages_local.sh` directly without `sudo`: +- **Ubuntu job**: runs in the `ubuntu:22.04` container +- **CentOS/manylinux job**: runs in the `manylinux_2_28_x86_64` container + +### Environment Variables + +| Variable | Default | Description | +|----------|---------|-------------| +| `ROCM_VERSION` | _(unset)_ | If set, selects tarball by **format** (see below). If unset locally: **latest nightly**. If unset in CI: channel selects latest **nightly** vs **release X.Y.Z**. | +| `ROCM_SDK_RELEASE_URL` | `https://repo.amd.com/rocm/tarball/` | HTML listing for **release** tarballs (`therock-dist-linux--X.Y.Z.tar.gz`). Used when `ROCM_SDK_CHANNEL=release` or `auto` with this URL set. | +| `ROCM_SDK_RELEASE_BASE_URL` | `https://repo.amd.com/rocm/tarball` | Directory URL for downloading **X.Y.Z** tarballs; overridden when version string is nightly-shaped. | +| `ROCM_SDK_BASE_URL` | See script | Effective tarball base after channel + version-shape resolution. | +| `ROCM_SDK_INDEX_URL` | `https://nightly.repo.amd.com/rocm/core/tarball/` | **Nightly** listing for latest nightly SDK discovery. Auto-fetch matches **SDK tarballs only** (`therock-dist-linux--.tar.gz`); `-tests-` entries are excluded **per filename** (version must follow `GPU_FAMILY-` immediately), not by filtering whole HTML lines. | +| `ROCM_SDK_NIGHTLY_BASE_URL` | `https://nightly.repo.amd.com/rocm/core/tarball` | Tarball base for **nightly** builds (`x.y.za…` versions). | +| `ROCM_SDK_NIGHTLY_INDEX_URL` | _(same as index default)_ | Optional override for nightly listing URL. | +| `ROCM_SDK_CHANNEL` | `auto` locally | **`nightly`** / **`release`** / **`auto`**. CI sets channel per trigger (see table above); manual with a pin uses **`auto`** so tarball follows version format. | +| `GPU_FAMILY` | `gfx110X-all` | ROCm SDK **tarball** family (`therock-dist-linux--…`); does not set TransferBench offload archs | +| `GPU_TARGETS` / `TRANSFERBENCH_GPU_TARGETS` | 13-arch default (see [`docs/transferbench.md`](../../docs/transferbench.md)) | HIP offload archs for the bundled TransferBench CLI when `BUILD_TRANSFERBENCH_CLI=ON` | +| `BUILD_TYPE` | `Release` | CMake build type (Release/Debug) | +| `BUILD_TRANSFERBENCH_CLI` | `OFF` locally; `ON` in CI | Build and install the TransferBench CLI in packages (`-DBUILD_TRANSFERBENCH_CLI`). CI uses `vars.BUILD_TRANSFERBENCH_CLI` or defaults to `ON`; `workflow_dispatch` input **`build_transferbench_cli`** overrides manual runs. | +| `ROCM_LIBPATCH_VERSION` | Auto-extracted from `ROCM_VERSION` | Major.minor in xxyy format with zero padding (e.g., `7.11` → `0711`, `8.0` → `0800`) - used for RVS version tagging | +| `CPACK_DEBIAN_PACKAGE_RELEASE` | Auto-generated | **Default** (`schedule`, `push`, `workflow_dispatch`, local): `r.` (e.g. `r0711.20260423` where `0711` = ROCm 7.11 from `ROCM_VERSION`). **Pull requests**: `r...`. **Release branches** (name starts with `rel`, non-PR): `GITHUB_RUN_NUMBER` (fallback: `1`). | +| `CPACK_RPM_PACKAGE_RELEASE` | same as `CPACK_DEBIAN_PACKAGE_RELEASE` | Identical to DEB. | +| `GITHUB_RUN_NUMBER` | `1` (local) | GitHub Actions run number - automatically set in CI, defaults to `1` for local builds | + +## Build Matrix + +The workflow builds packages for: + +| Platform | Container/Runner | Package Types | Script Mode | +|----------|------------------|---------------|-------------| +| Ubuntu 22.04 | `ubuntu:22.04` container on `ubuntu-22.04` host | DEB, TGZ | Auto-detects Ubuntu | +| Manylinux 2.28 (AlmaLinux 8) | `manylinux_2_28_x86_64` container | RPM, TGZ | Auto-detects AlmaLinux | + +## Package Naming Convention + +Packages are named automatically by CPack using the RVS version from `CMakeLists.txt`: + +- **DEB**: `amdrocm-rvs_${RVS_VERSION}_amd64.deb` +- **RPM**: `amdrocm-rvs-${RVS_VERSION}.x86_64.rpm` +- **TGZ**: `amdrocm-rvs---Linux.tar.gz` (CPack: same **release** suffix as DEB/RPM, e.g. `r0711.20260423` from `ROCM_LIBPATCH_VERSION` and date, or a PR-specific suffix) + +The **patch version** in `RVS_VERSION` is automatically computed from the number of commits since the last `v..*` git tag. For example, with tag `v1.3.0` and 15 commits since, and a release of `r0711.20260423`: +``` +amdrocm7-rvs_1.3.15-r0711.20260423_amd64.deb +amdrocm7-rvs-1.3.15-r0711.20260423.el8.x86_64.rpm +amdrocm7-rvs-1.3.15-r0711.20260423-Linux.tar.gz +``` + +`CPACK_PACKAGE_FILE_NAME` in CMake is set to include the same **release** as DEB/RPM (from `CPACK_RPM_PACKAGE_RELEASE` in the build environment). + +If no matching `v` tag is found, the patch defaults to `0` from `project(VERSION)` in `CMakeLists.txt`. + +**Important**: Package filenames match the internal package metadata, ensuring compliance with Debian and RPM standards. The version major and minor are sourced from `CMakeLists.txt` via CMake's `project(VERSION)` command, while the patch is auto-computed from git history at configure time. + +## Installing Generated Packages + +### Ubuntu/Debian (DEB) + +```bash +# Download the package from S3 (or use the apt repo described above) +sudo dpkg -i amdrocm7-rvs_*.deb + +# Run RVS +/opt/rocm/extras-7/bin/rvs --help +``` + +### CentOS/RHEL/Rocky Linux (RPM) + +```bash +# Download the package from S3 (or use the yum/dnf repo described above) +sudo rpm -i --replacefiles --nodeps amdrocm7-rvs-*.rpm + +# Run RVS +/opt/rocm/extras-7/bin/rvs --help +``` + +### Any Linux Distribution (TGZ - Relocatable) + +#### Pre-install: ROCm + +Before you install the RVS TGZ, **ROCm must be installed, configured, and on your `PATH` / `LD_LIBRARY_PATH` as in AMD’s documentation** so HIP, HSA, and other ROCm libraries are discoverable. Follow the current Linux install guide in **rocm docs**: + +- **ROCm documentation (start here)**: +- **Linux install / deployment (paths, env, post-install)**: + +**Assumptions (TGZ use):** ROCm is set up on the machine, `ROCM_PATH` (or your install prefix) is correct, and the runtime can load ROCm libraries. Install a ROCm stack that includes **ROCm’s LLVM** (for example the **`rocm-llvm`** package from the ROCm repo). **DEB/RPM** packages from this project declare that dependency; TGZ users should mirror that on the host. The TGZ only ships RVS; it does not replace a full ROCm stack. + +#### Install RVS run-time dependencies (on the target system) + +The TGZ is built against ROCm; on the target host you still need **PCI** for GPU enumeration and **NUMA** when using the bundled **TransferBench** CLI: + +| Family | Typical PCI package | NUMA (TransferBench CLI) | +|--------|---------------------|---------------------------| +| **Debian / Ubuntu** | `libpci3` (or `libpci-3-0-0` on some releases) | `libnuma1` | +| **RHEL / Rocky / Alma 8+** | `pciutils-libs` | `numactl-libs` | +| **SUSE / openSUSE** | `libpci3` / `pciutils` as appropriate for the release | `libnuma1` | + +```bash +# Ubuntu / Debian +sudo apt update && sudo apt install -y libpci3 libnuma1 + +# RHEL / Rocky / Alma 8+ +sudo dnf install -y pciutils-libs numactl-libs + +# openSUSE / SUSE (adjust package names per release) +sudo zypper install libpci3 libnuma1 +``` + +User-facing install steps for TGZ (PATH / `LD_LIBRARY_PATH`, **`RPATH`** including **`/opt/rocm/lib`**, **`/opt/rocm/lib/llvm/lib`**, **`/opt/rocm/core-/lib`**, and **`/opt/rocm/core-/lib/llvm/lib`**) are in **[docs/INSTALL_TGZ.md](../../docs/INSTALL_TGZ.md)**. + +#### Extract the TGZ + +```bash +# Extract to the extras directory (example for ROCm major 7; match your path) +sudo mkdir -p /opt/rocm/extras-7 +sudo tar -xzf amdrocm7-rvs-*.tar.gz -C /opt/rocm/extras-7 +``` + +#### Post-install: `PATH` and `LD_LIBRARY_PATH` + +Point the shell at the extracted RVS prefix and the ROCm you installed (replace paths with your real `ROCM_PATH` and extras major version). Copy and paste the block as one unit: + +```bash +export ROCM_PATH=/opt/rocm # or your real ROCm root, per rocm docs +export PATH=/opt/rocm/extras-7/bin:$ROCM_PATH/bin:$PATH +export LD_LIBRARY_PATH=/opt/rocm/extras-7/lib:$ROCM_PATH/lib:$ROCM_PATH/lib/llvm/lib:$LD_LIBRARY_PATH +``` + +**`rvs`** embeds **`RPATH`** for **`/opt/rocm/lib`**, **`/opt/rocm/lib/llvm/lib`**, **`core-`** paths, and the extras-relative **`$ORIGIN`** paths, so it usually does not need **`LD_LIBRARY_PATH`** for them. The export still includes **`$ROCM_PATH/lib/llvm/lib`** for a typical ROCm tree, other tools, and troubleshooting; if LLVM is only under **`$ROCM_PATH/core-/lib/llvm/lib`**, add that path instead or as well. + +**Run RVS** + +```bash +rvs --help +``` + +## Verifying Packages + +The workflow automatically verifies package contents: + +### DEB Package Verification + +```bash +dpkg-deb -I amdrocm*-rvs_*.deb # Package info +dpkg-deb -c amdrocm*-rvs_*.deb # Package contents +``` + +### RPM Package Verification + +```bash +rpm -qip amdrocm*-rvs-*.rpm # Package info +rpm -qlp amdrocm*-rvs-*.rpm # Package contents +rpm -qRp amdrocm*-rvs-*.rpm # Package dependencies +``` + +## Accessing Build Packages + +Packages are uploaded directly to **S3** (not GitHub Actions artifacts). To find them: + +1. Go to the **Actions** tab in your GitHub repository +2. Click on the latest workflow run +3. Download the **`build-report`** artifact — it contains S3 console links to each package location +4. Or browse S3 directly using the path layout described above + +## Customization + +### Changing ROCm Version or GPU Family + +**Option 1: Via GitHub Actions UI (Manual Trigger)** + +1. Go to **Actions** → **Build Relocatable Packages** +2. Click **Run workflow** +3. Enter custom values for: + - ROCm Version (e.g., `7.11.0a20260121`) + - GPU Family (e.g., `gfx110X-all`) + +**Option 2: Edit Workflow Defaults** + +Edit the `env` section in `.github/workflows/build-relocatable-packages.yml`: + +```yaml +env: + ROCM_VERSION: '7.11.0a20260121' # Change this + GPU_FAMILY: 'gfx110X-all' # Change this + BUILD_TYPE: Release +``` + +**Option 3: Edit Build Script Defaults** + +Edit `build_packages_local.sh`: + +```bash +ROCM_VERSION="${ROCM_VERSION:-7.11.0a20260121}" # Change default here +GPU_FAMILY="${GPU_FAMILY:-gfx110X-all}" # Change default here +BUILD_TYPE="${BUILD_TYPE:-Release}" +``` + +### Adding More Distributions + +To add support for more distributions, extend the workflow matrix: + +**For Ubuntu variants:** + +```yaml +build-ubuntu: + strategy: + matrix: + ubuntu_version: ['20.04', '22.04', '24.04'] + runs-on: ubuntu-${{ matrix.ubuntu_version }} +``` + +**For other RPM-based distributions:** + +```yaml +build-fedora: + runs-on: ubuntu-latest + container: + image: fedora:39 + steps: + - uses: actions/checkout@v4 + - run: | + chmod +x build_packages_local.sh + ROCM_VERSION=${{ env.ROCM_VERSION }} \ + GPU_FAMILY=${{ env.GPU_FAMILY }} \ + ./build_packages_local.sh +``` + +**For SLES/OpenSUSE:** + +```yaml +build-sles: + runs-on: ubuntu-latest + container: + image: opensuse/leap:15.5 + steps: + - uses: actions/checkout@v4 + - run: | + chmod +x build_packages_local.sh + ROCM_VERSION=${{ env.ROCM_VERSION }} \ + GPU_FAMILY=${{ env.GPU_FAMILY }} \ + ./build_packages_local.sh +``` + +Note: You may need to update the OS detection logic in `build_packages_local.sh` to handle additional distributions. + +### Customizing Build Parameters + +Edit the CMake arguments in `build_packages_local.sh`, or override **`CMAKE_INSTALL_RPATH`** / **`CMAKE_SKIP_RPATH`** from the command line when needed. Defaults match packaged builds: + +```bash +cmake -B "$BUILD_DIR" \ + -DCMAKE_BUILD_TYPE="$BUILD_TYPE" \ + -DROCM_PATH="$ROCM_PATH" \ + -DHIP_PLATFORM=amd \ + -DROCM_MAJOR_VERSION="$ROCM_MAJOR" \ + -DCMAKE_INSTALL_PREFIX="/opt/rocm/extras-${ROCM_MAJOR}" \ + -DCPACK_PACKAGING_INSTALL_PREFIX="/opt/rocm/extras-${ROCM_MAJOR}" \ + -DCMAKE_VERBOSE_MAKEFILE=1 \ + -DFETCH_ROCMPATH_FROM_ROCMCORE=ON \ + -DYOUR_CUSTOM_OPTION=ON # Add custom options here +``` + +## Troubleshooting + +### S3: "Credentials could not be loaded" + +- **PR from a fork:** S3 upload is skipped for fork PRs (secrets are not passed). Use a branch in the same repo or push to `main`/`master` to upload. +- **Same repo / push:** Ensure the `AWS_ROLE_ARN` secret is set (Settings → Secrets and variables → Actions → Secrets) and the IAM role’s trust policy allows GitHub OIDC for this repo. Ensure the OIDC identity provider exists in the AWS account (`token.actions.githubusercontent.com`). + +### Package Build Fails + +1. Check the workflow logs in GitHub Actions +2. Verify the ROCm tarball URL is accessible +3. Run `./build_packages_local.sh` locally to reproduce the issue +4. Check that all dependencies were installed correctly +5. Review CMake configuration output + +### RPATH Issues + +If binaries can't find libraries: + +```bash +# Check RUNPATH on an installed package binary (replace 7 with your ROCm major version) +readelf -d /opt/rocm/extras-7/bin/rvs | grep -E 'RUNPATH|RPATH' +patchelf --print-rpath /opt/rocm/extras-7/bin/rvs + +# Packaged binaries should only use $ORIGIN* and /opt/rocm/* entries (no build SDK paths). +# Local build-tree binaries may still list $ROCM_PATH from the build host — that is expected. +``` + +### Missing Dependencies + +If the package reports missing dependencies: + +```bash +# Check what libraries are needed (replace 7 with your ROCm major version) +ldd /opt/rocm/extras-7/bin/rvs + +# Install missing ROCm components if needed +``` + +### Local Testing + +To test the workflow locally before pushing: + +```bash +# Option 1: Auto-fetch latest ROCm version +sudo ./build_packages_local.sh + +# Option 2: Set environment variables for custom configuration +export ROCM_VERSION=7.11.0a20260121 +export GPU_FAMILY=gfx110X-all +export BUILD_TYPE=Release + +# Run with sudo -E to preserve environment variables +sudo -E ./build_packages_local.sh + +# Check generated packages +ls -lh build/amdrocm*-rvs* +``` + +## References + +- [TheRock Releases Documentation](https://github.com/ROCm/TheRock/blob/main/RELEASES.md) +- [TheRock Nightly Tarballs](https://therock-nightly-tarball.s3.amazonaws.com/index.html) +- [Local Build Script](../build_packages_local.sh) - Core build engine used by workflow +- [Quick Start Guide](../QUICKSTART_PACKAGES.md) - Step-by-step local build instructions +- [Package Build Summary](../PACKAGE_BUILD_SUMMARY.md) - Technical overview and architecture +- [RVS Build Instructions](../README.md) +- [ROCm Documentation](https://rocm.docs.amd.com/) + +## Support + +For issues with: +- **RVS Build**: Open an issue in this repository +- **ROCm SDK**: See [TheRock Issues](https://github.com/ROCm/TheRock/issues) +- **Workflow**: Check GitHub Actions documentation + +## License + +This workflow is part of ROCm Validation Suite and follows the same MIT license. diff --git a/.github/workflows/README_NIGHTLY_TESTS.md b/.github/workflows/README_NIGHTLY_TESTS.md new file mode 100644 index 000000000..60d791c3d --- /dev/null +++ b/.github/workflows/README_NIGHTLY_TESTS.md @@ -0,0 +1,489 @@ +# RVS Nightly Tests Workflow + +This document describes [`.github/workflows/rvs-nightly-tests.yml`](./rvs-nightly-tests.yml), +which picks up the **latest RVS tarball** from the tarball index (`secrets.RVS_TARBALL_INDEX_URL`, passed through workflow `env.TARBALL_INDEX_URL` — **no default URL in the repo**), after a successful **Build Relocatable Packages** run (or on manual dispatch), copies it to a **configurable remote target node** over SSH, +installs it there, and runs RVS level 4 on that node. + +The GitHub Actions runner ("RVS Runner") is only an **orchestrator** — it +doesn't need a GPU or ROCm installed locally. All RVS install, binary +verification, and `rvs -r 4` execution happens on the target +node. The target node is configurable so the same workflow can be pointed +at any GPU host without code changes. + +## What it does + +``` +workflow_run (push main|master or scheduled build) / manual + │ + ▼ +install-rvs-on-target [self-hosted orchestrator] ──ssh──▶ [target GPU node] + │ resolve index + validate paths (orchestrator only) + │ setup-ssh → download .tar.gz on runner (curl never on target) → scp to target + │ verify-rocm → install-rvs → verify-rvs-binary (commands run on target via ssh) + ▼ +run-rvs-level-4 [self-hosted orchestrator] ──ssh──▶ [target GPU node] + │ rvs_nightly_test.sh: run-level4, collect-logs, capture-versions + │ upload intermediate logs artifact; cleanup remote work dir + ▼ +create-test-report [utility runner] + │ rvs_nightly_test.sh build-report → SUMMARY.md + final artifact + ▼ +artifact: rvs-nightly-report- +``` + +Install, test, and report logic lives in [`rvs_nightly_test.sh`](../../rvs_nightly_test.sh) +at the repo root — the same split as `build-relocatable-packages.yml` + +`build_packages_local.sh`. + +## Triggers + +| Trigger | Cadence | What fires | +|---|---|---| +| `workflow_run` | After **Build Relocatable Packages** completes | Runs only when that workflow's overall conclusion is **success** and the triggering event was either **`push`** to **`main`** / **`master`**, or the package workflow's own **`schedule`** (daily build). PR builds, release-branch builds, failed builds, and manual package builds do **not** start nightly tests. | +| `workflow_dispatch` | Manual | Always runs. Supports overriding the tarball URL and **retargeting at any node** without editing the workflow. | + +Nightly tests no longer have their own cron — they follow successful **Build Relocatable Packages** runs (see that workflow's `0 13 * * *` UTC schedule for the daily package build cadence). + +## Manual dispatch inputs + +| Input | Default | Description | +|---|---|---| +| `tarball_url` | _(empty)_ | If set, the workflow downloads this exact URL instead of scraping the index. Useful for re-running an older build. | +| `target_node` | _(empty → `secrets.RVS_TARGET_NODE`)_ | Hostname or IP of the node to install RVS on and run tests against. **This is the value that retargets the test execution.** Stored as a secret so the lab node identity isn't visible in repo settings or run logs (GitHub Actions automatically masks secret values as `***` in step output). | +| `target_user` | _(empty → `secrets.RVS_TARGET_USER`; if both are unset, SSH defaults to the orchestrator runner's local user)_ | SSH user on the target node. Must have `NOPASSWD` sudo on the target (see prerequisites). Stored as a secret so the lab account name isn't visible in repo settings or run logs (GitHub Actions automatically masks secret values as `***` in step output). | +| `remote_work_dir` | _(empty → `vars.RVS_REMOTE_WORK_DIR`, then `/tmp/rvs-nightly-`)_ | Working dir on the target node where the tarball is staged, logs are written, and which gets `rm -rf`'d at the end. | +| `target_rocm_path` | _(empty → `vars.RVS_TARGET_ROCM_PATH`)_ — **required**, no hard-coded default | Absolute path to the ROCm tarball install root on the target node. This is the directory containing `bin/rocminfo`, `bin/amd-smi`, `lib/`, `lib/llvm/lib/`, and `lib/rocm_sysdeps/lib/` (the layout produced by the ROCm tarball install method). The workflow fails fast in the validate step if neither the input nor the variable is set. | + +Workflow inputs **win over repo variables**, so individual `workflow_dispatch` +runs can be retargeted from the Actions UI without changing repo settings. + +Example — point a single run at a specific node: + +```bash +gh workflow run rvs-nightly-tests.yml \ + -f tarball_url="/amdrocm7-rvs-1.4.21-288-Linux.tar.gz" \ + -f target_node="" \ + -f target_user="" +``` + +Example — resolve from your index secret (typical manual run; ensure `RVS_TARBALL_INDEX_URL` is set for post-build runs): + +```bash +gh workflow run rvs-nightly-tests.yml -f target_node="" +``` + +Example — point at a specific ROCm install on a multi-ROCm host: + +```bash +gh workflow run rvs-nightly-tests.yml \ + -f target_node="" \ + -f target_rocm_path="" +``` + +## Repository configuration + +**Variables** (Settings → Secrets and variables → Actions → Variables): + +| Name | Required? | Purpose | +|---|---|---| +| `RVS_NIGHTLY_INDEX_RUNNER_LABEL` | optional (defaults to `RVS_TEST_RUNNER_LABEL`) | Runner label for **install-rvs-on-target** (resolve + download + scp). Use a self-hosted pool with HTTPS egress to the tarball index and `.tar.gz` host. The GPU target never performs those downloads. If unset, `RVS_TEST_RUNNER_LABEL` (then `self-hosted`) is used. | +| `RVS_REMOTE_WORK_DIR` | optional (default `/tmp/rvs-nightly-`) | Working dir on the target node. Cleared with `rm -rf` at the end of the job. | +| `RVS_TARGET_ROCM_PATH` | **Required** *(unless every run sets `target_rocm_path` input)* | Absolute path to the ROCm tarball install root on the target node — the directory that contains `bin/rocminfo`, `bin/amd-smi`, `lib/`, `lib/llvm/lib/`, and `lib/rocm_sysdeps/lib/`. The workflow doesn't assume any conventional path (no `/opt/rocm` default), since tarball installs land wherever you extracted them. | +| `RVS_TEST_RUNNER_LABEL` | optional (default `self-hosted`) | Label for **run-rvs-level-4**. **install-rvs-on-target** uses `RVS_NIGHTLY_INDEX_RUNNER_LABEL` when set, otherwise this label — that job resolves the index, **downloads the tarball on the orchestrator**, and `scp`s it to the target (the target never curls the CDN). Requires `ssh`, `scp`, `curl`, and HTTPS egress to the tarball hosts. Does not need a GPU or ROCm. | + +**Secrets** (Settings → Secrets and variables → Actions → Secrets): + +| Name | Required? | Purpose | +|---|---|---| +| `RVS_TARBALL_INDEX_URL` | **Required** *(unless every run that needs an index provides `workflow_dispatch` `tarball_url`; `workflow_run` needs this secret)* | HTTPS directory listing URL scraped for the latest `amdrocm*-rvs-*-Linux.tar.gz` (e.g. `https:///nightly/rvs/tar/`). Copied into workflow `env.TARBALL_INDEX_URL`. GitHub masks secret values in logs when printed. If this name was previously an Actions **Variable**, copy the value to this secret and remove the variable. | +| `RVS_TARGET_NODE` | **Required** *(unless every run sets `target_node` input)* | Hostname or IP of the node where RVS is installed and tests run. Stored as a secret so the lab node identity isn't visible in repo Variables or in run logs — GitHub Actions automatically masks secret values as `***` wherever they appear in step output. Workflow fails fast if neither this secret nor `target_node` input is set. | +| `RVS_TARGET_USER` | optional (no hard-coded default) | SSH user on the target node. If unset and `target_user` input is empty, the SSH client falls back to the orchestrator runner's local user — set this secret explicitly to avoid surprises. Stored as a secret so the lab account name isn't visible in repo settings or run logs (auto-masked as `***`). | +| `RVS_TARGET_SSH_KEY` | **Required** | Private SSH key (OpenSSH or PEM format) authorized on the target node for `RVS_TARGET_USER`. Written to `$RUNNER_TEMP/rvs_target_key` for the duration of the job and scrubbed in the cleanup step. | + +## How the latest tarball is picked + +The **Resolve latest tarball URL** step at the start of **install-rvs-on-target** runs on the orchestrator with `$INDEX_URL` = `TARBALL_INDEX_URL` from the workflow env (set from `secrets.RVS_TARBALL_INDEX_URL`), unless `workflow_dispatch.tarball_url` is set: + +```bash +curl -sL "$INDEX_URL" \ + | grep -oE 'amdrocm[0-9]*-rvs-[0-9A-Za-z._\-]+-Linux\.tar\.gz' \ + | sort -uV \ + | tail -n 1 +``` + +The regex matches any `amdrocm-rvs-…-Linux.tar.gz` filename in the +directory-listing HTML. `sort -V` is GNU "version sort" so version +suffixes like `1.4.21-9` and `1.4.21-100` compare correctly. The +**lexicographically largest by version** is selected. + +## How the tarball is installed (on the target node) + +The runner derives the ROCm major version from the tarball filename +(e.g. `amdrocm7-rvs-1.4.21-…-Linux.tar.gz` → `7`), combines it with +`TARGET_ROCM_PATH` (which selects *which* ROCm install to use), writes +the derived paths to `$GITHUB_ENV`, then SSHes into the target with those +values exported so the install runs against the matching `extras-` +directory under the chosen ROCm. The same workflow handles ROCm 6, 7, +etc., and any version inside `7.x` without code changes: + +```bash +# On the orchestrator (Validate configuration step in install-rvs-on-target): +# Parsed from $TARBALL_NAME via [[ "$TARBALL_NAME" =~ ^amdrocm([0-9]+)- ]] +ROCM_MAJOR=7 +# INSTALL_DIR is always /opt/rocm/extras-/ — it's where the RVS +# tarball gets extracted. This is decoupled from TARGET_ROCM_PATH so the +# install location matches the manual command verbatim regardless of +# which ROCm install the workflow is told to run *against*. +INSTALL_DIR=/opt/rocm/extras-${ROCM_MAJOR} # /opt/rocm/extras-7 +RVS_BIN=${INSTALL_DIR}/bin/rvs + +# TARGET_ROCM_PATH (from inputs.target_rocm_path / vars.RVS_TARGET_ROCM_PATH; +# required, no default) is *where the ROCm runtime libraries live*. Set +# the repo variable to the absolute install root of the ROCm tarball you +# want RVS to run against (the directory that contains bin/, lib/, etc.). + +# On the target node (Install RVS on target node step), via SSH: +# sudo is used only when $INSTALL_DIR isn't user-writable. /opt/rocm/* is +# typically root-owned so this picks up sudo -n automatically. +mkdir -p "$INSTALL_DIR" # (with `sudo -n` if needed) +tar -xzf "$REMOTE_WORK_DIR/pkg/.tar.gz" -C "$INSTALL_DIR" + +# LD_LIBRARY_PATH is wired off INSTALL_DIR (for RVS's own libs) plus the +# three TheRock-style subdirs of TARGET_ROCM_PATH (matches the official +# ROCm tarball-install docs). This is what makes a non-/opt/rocm +# TARGET_ROCM_PATH actually do something useful. +export LD_LIBRARY_PATH="${INSTALL_DIR}/lib:${TARGET_ROCM_PATH}/lib/rocm_sysdeps/lib:${TARGET_ROCM_PATH}/lib/llvm/lib:${TARGET_ROCM_PATH}/lib:${LD_LIBRARY_PATH:-}" + +"$RVS_BIN" --version +``` + +### Install location vs. ROCm runtime path + +Two paths in the workflow look similar but mean very different things, and +keeping them straight is what made the multi-ROCm-host case finally work: + +| Variable | What it is | Default | How to override | +|---|---|---|---| +| `INSTALL_DIR` | Where the RVS tarball gets extracted on the target node. Always under `/opt/rocm/`, matching the manual command. | `/opt/rocm/extras-${ROCM_MAJOR}` | Not configurable by design — the install location is canonical, decoupled from where ROCm itself lives. | +| `TARGET_ROCM_PATH` | Where the ROCm runtime libraries live (the directory containing `bin/`, `lib/`, `lib/llvm/lib/`, `lib/rocm_sysdeps/lib/`). Drives `LD_LIBRARY_PATH`, the prereq-check probe, and the version row in the report. | **(no default — required)** | `inputs.target_rocm_path` (workflow_dispatch) or `vars.RVS_TARGET_ROCM_PATH` (repo Variables tab). Workflow fails fast in the validate step if neither is set. | + +This split exists because hosts that have multiple ROCm installs side-by-side +(or that installed ROCm via a TheRock tarball under `$HOME`) almost never +keep the runtime libs at `/opt/rocm/lib/`. The workflow needs to know where +`/lib/`, `/lib/llvm/lib/`, and `/lib/rocm_sysdeps/lib/` actually are; the +install location of RVS itself should not — and does not — care. + +The runtime `LD_LIBRARY_PATH` for every step that executes `rvs` (install, +ldd-verify, level 4, report `--version` query) is now: + +```bash +LD_LIBRARY_PATH=$INSTALL_DIR/lib:$TARGET_ROCM_PATH/lib/rocm_sysdeps/lib:$TARGET_ROCM_PATH/lib/llvm/lib:$TARGET_ROCM_PATH/lib:$LD_LIBRARY_PATH +``` + +Mapping back to the manual sequence the RVS team uses today: + +```bash +sudo mkdir -p /opt/rocm/extras-7 +sudo tar -xzf amdrocm7-rvs-1.4.21-288-Linux.tar.gz -C /opt/rocm/extras-7 +export LD_LIBRARY_PATH=/opt/rocm/extras-7/lib:/install/lib:/install/lib/rocm_sysdeps/:/install/lib/llvm/lib:$LD_LIBRARY_PATH +``` + +| Manual entry | Workflow equivalent | Notes | +|---|---|---| +| `/opt/rocm/extras-7/lib` | `${INSTALL_DIR}/lib` | Same path — `INSTALL_DIR` is literally `/opt/rocm/extras-`. | +| `/install/lib` | `${TARGET_ROCM_PATH}/lib` | The manual command's `/install/` was a literal filesystem path that only existed on the RVS build container. The workflow replaces it with the configurable `TARGET_ROCM_PATH/lib`, which is what "the ROCm install's lib directory" actually means on a real host. | +| `/install/lib/rocm_sysdeps/` | `${TARGET_ROCM_PATH}/lib/rocm_sysdeps/lib` | Same fix, plus the trailing `/lib` that the official ROCm tarball docs include and the manual command was missing. | +| `/install/lib/llvm/lib` | `${TARGET_ROCM_PATH}/lib/llvm/lib` | Where `libomp.so` lives in TheRock builds. | + +### What to set `RVS_TARGET_ROCM_PATH` to + +Set it to the absolute path of the ROCm **tarball install root** on the target node — i.e. the directory you (or whoever provisioned the node) chose when extracting the ROCm distribution tarball. That directory must contain at least `bin/rocminfo`, `bin/amd-smi`, `lib/`, `lib/llvm/lib/`, and `lib/rocm_sysdeps/lib/`. There is no hard-coded default; the workflow fails fast in the validate step if `RVS_TARGET_ROCM_PATH` and the per-run `target_rocm_path` input are both empty. + +The RVS tarball itself still always lands in `/opt/rocm/extras-/` regardless — that part is invariant. + +### TheRock-style tarball ROCm installs + +The [official ROCm 7.12+ "tarball" install method](https://rocm.docs.amd.com/en/7.12.0-preview/install/rocm.html?fam=instinct&gpu=mi350x&os=ubuntu&os-version=24.04&i=tar) +doesn't drop ROCm under `/opt/rocm-/`. Instead you extract a +single distribution archive (e.g. +`therock-dist-linux--dcgpu-.tar.gz`) into an +arbitrary directory, set `ROCM_PATH=$(pwd)/install`, and source the +env. So the layout looks like: + +``` +/home///install/ +├── bin/ ← rocm-smi, rocminfo, hipcc, … +├── lib/ +│ ├── rocm_sysdeps/lib/ ← ROCm runtime libs live HERE in TheRock builds +│ └── llvm/lib/ +├── share/ +└── … +``` + +This works with the workflow as-is, with two things to be aware of: + +1. **Find the absolute path of the install on the target node** — there's no canonical location, you pick it at install time. Either ask whoever installed it, or search: + ```bash + ssh @ ' + for cand in ~/install ~/*/install ~/rocm*/install /opt/therock*/install; do + [ -x "$cand/bin/rocminfo" ] && echo " found ROCm install: $cand" + done + ' + ``` +2. **Pass that absolute path as `target_rocm_path`.** Example: + ```bash + gh workflow run rvs-nightly-tests.yml \ + -f target_node= \ + -f target_rocm_path=$HOME//install + ``` + +What the workflow handles automatically for tarball installs: + +- The install step **skips `sudo`** when the destination is user-writable (TheRock installs in `$HOME` don't need root); only system `/opt/rocm-*` installs trigger `sudo -n`. +- The prereq check invokes `${TARGET_ROCM_PATH}/bin/rocminfo` and `${TARGET_ROCM_PATH}/bin/amd-smi version` directly by absolute path. TheRock tarball binaries have RPATH/RUNPATH baked in relative to `${TARGET_ROCM_PATH}/lib`, so they resolve their own ROCm libs without needing `PATH` or `LD_LIBRARY_PATH` setup. +- The version-string row in the report uses what `rvs --version` prints, so it always reflects the RVS tarball — not the underlying ROCm. The major-version cross-check the workflow used to do against `.info/version` is gone (TheRock installs don't ship one). + +**To use a TheRock-style install just set `vars.RVS_TARGET_ROCM_PATH` (or pass `-f target_rocm_path=...`) to the absolute install root, e.g. `$HOME//install`. The workflow's `LD_LIBRARY_PATH` is wired off `TARGET_ROCM_PATH` (see [Install location vs. ROCm runtime path](#install-location-vs-rocm-runtime-path)), so the loader picks up `lib/`, `lib/llvm/lib/`, and `lib/rocm_sysdeps/lib/` from the install you point at — no file edits required. + +`sudo -n` is non-interactive — it fails fast instead of hanging if +`NOPASSWD` isn't configured. There's no PTY over the SSH channel anyway, +so an interactive sudo prompt would deadlock the job. + +The step fails fast if the filename doesn't match `^amdrocm-`, or +if `$RVS_BIN` isn't executable after extraction. + +### Prerequisites + +**On the GitHub runner (orchestrator):** + +- `ssh`, `scp`, `ssh-keyscan`, and `curl` on `PATH`. +- **install-rvs-on-target** runs on `vars.RVS_NIGHTLY_INDEX_RUNNER_LABEL` or `vars.RVS_TEST_RUNNER_LABEL` (see Variables). On that runner only: `curl` resolves the latest tarball from the index, `curl` downloads the `.tar.gz` to `./pkg/`, then `scp` pushes it to the target. The GPU target never curls the index or CDN. +- Network egress from that orchestrator to: + - the tarball index and tarball file hosts (HTTPS port 443; from `secrets.RVS_TARBALL_INDEX_URL` via workflow `env.TARBALL_INDEX_URL` unless `workflow_dispatch.tarball_url` is set), + - the target node (SSH, typically port 22). +- The runner does **not** need a GPU, ROCm, or `sudo`. + +**On the target node** (all enforced by the **Pre-flight ROCm checks** below — the workflow fails fast if any are missing): + +- SSH server reachable from the runner, with the public counterpart of `secrets.RVS_TARGET_SSH_KEY` authorized for `$TARGET_USER`. +- `$TARGET_USER` has **`NOPASSWD` sudo** for `mkdir` + `tar` into `/opt/rocm/extras-` — the install step uses `sudo -n` (when the path isn't user-writable) and aborts otherwise. +- A ROCm tarball install at `$TARGET_ROCM_PATH` (configured via `vars.RVS_TARGET_ROCM_PATH` or the per-run `target_rocm_path` input) that provides at least `bin/rocminfo` and `bin/amd-smi`. The RVS tarball only ships RVS, not the rest of ROCm, so the runtime libs under `$TARGET_ROCM_PATH/lib`, `$TARGET_ROCM_PATH/lib/llvm/lib`, and `$TARGET_ROCM_PATH/lib/rocm_sysdeps/lib` must be present. +- Working kernel driver (`amdgpu`) and at least one GPU enumerated by `rocminfo` / `amd-smi`. + +## Pre-flight ROCm checks + +Before extracting the RVS tarball on the target, the workflow runs **`Verify ROCm prerequisites on target node`** over SSH. It's deliberately minimal — two probes against the actual install at `$TARGET_ROCM_PATH`: + +| Sub-check | Action on failure | Catches | +|---|---|---| +| `$TARGET_ROCM_PATH` directory exists | hard fail (`exit 1`) | ROCm not installed on target, or the configured path is wrong (typo, missing trailing component, etc.) | +| `$TARGET_ROCM_PATH/bin/rocminfo` exits 0 | hard fail (via `set -euo pipefail`) | Driver not loaded, no GPU exposed, runtime libs under `$TARGET_ROCM_PATH/lib` missing/broken, group permissions wrong | +| `$TARGET_ROCM_PATH/bin/amd-smi version` exits 0 | hard fail | `amd-smi` missing from the install, ROCm SMI library mismatch, or driver/runtime broken | + +Both binaries are invoked by absolute path (`${TARGET_ROCM_PATH}/bin/...`), so the workflow doesn't depend on `PATH` or `LD_LIBRARY_PATH` being set up — the binaries' baked-in RPATH/RUNPATH resolves the runtime libs from `${TARGET_ROCM_PATH}/lib`. If either probe fails, the step fails fast with the binary's own error message in the log, which is usually more diagnostic than anything the workflow could add on top. + +A typical successful log (with `target_rocm_path=`): + +``` +=== System === +Linux 6.x.x-x-generic #... SMP ... x86_64 GNU/Linux + +=== Target ROCm path: === + +=== /bin/rocminfo === +ROCk module version is loaded +HSA Agents +========== +Agent 1 + Name: AMD Instinct ... + ... +Agent 2..N: (one per GPU) + +=== /bin/amd-smi version === +AMDSMI Tool: | AMDSMI Library version: | ROCm version: + +::notice::ROCm prerequisites OK on target node at +``` + +After install, **`Verify RVS binary library resolution on target node`** runs `ldd "$RVS_BIN"` over SSH (where `$RVS_BIN` = `/opt/rocm/extras-${ROCM_MAJOR}/bin/rvs`) and **hard-fails** the job if any library shows up as `not found`. This prevents the workflow from spending hours on `rvs -r 4` only to discover a `dlopen` error in the level log. The full `ldd` output is printed for diagnostic purposes — useful for spotting which `$TARGET_ROCM_PATH/lib/...` paths each library resolved from. + +### Manual one-liner (validate a candidate target node) + +To verify a node is viable before pointing the workflow at it, SSH into the candidate and run the same two probes the workflow runs (substituting `` with what you plan to set `target_rocm_path` to): + +```bash +ROCM_PATH= +set -euo pipefail +[ -d "$ROCM_PATH" ] || { echo "$ROCM_PATH does not exist"; exit 1; } +"$ROCM_PATH/bin/rocminfo" +"$ROCM_PATH/bin/amd-smi" version +sudo -n true 2>/dev/null && echo "NOPASSWD sudo OK" || echo "warn: sudo requires password (install step will fail)" +echo "Target node OK" +``` + +If both `rocminfo` and `amd-smi version` exit zero, the workflow's prereq step will pass on this node. + +## The test + +Run verbatim on the target node (with `$RVS_BIN` = `/opt/rocm/extras-${ROCM_MAJOR}/bin/rvs`, +populated by the validate step — for an `amdrocm7-…` tarball this resolves +to `/opt/rocm/extras-7/bin/rvs`): + +```bash +"$RVS_BIN" -r 4 +``` + +The command is executed over SSH. The step captures full stdout/stderr to +`$REMOTE_WORK_DIR/reports/rvs_level_4.log` on the target, then the +**`Collect logs from target node`** step `scp`s the log file back to +`./reports/` on the runner. The command's exit code is propagated through +SSH and recorded in a step output, and the job is marked failed at the end +if level 4 exited non-zero. + +## Test report + +After RVS finishes and the log is collected, the `Build test report` +step generates `reports/SUMMARY.md` (also written to the GitHub job summary), +e.g.: + +```markdown +# RVS Nightly Test Report + +| Field | Value | +|---|---| +| Run | `1234567890` | +| Trigger | `workflow_run` (scheduled package build) | +| Target ROCm path | `` (version ``) | +| Remote work dir | `/tmp/rvs-nightly-1234567890` | +| Tarball | `amdrocm7-rvs-1.4.21-288-Linux.tar.gz` | +| Source URL | resolved tarball URL (from `workflow_dispatch.tarball_url` or index at `secrets.RVS_TARBALL_INDEX_URL`) | +| RVS version | `RVS 1.4.21.0-...` | +| Overall result | **PASS** | + +## Results + +| Test | Command | Result | Exit | Started (UTC) | Ended (UTC) | +|---------|--------------------------------------|:------:|-----:|-----------------------|-----------------------| +| Level 4 | `/opt/rocm/extras-7/bin/rvs -r 4` | PASS | 0 | 2026-05-19T15:25:12Z | 2026-05-19T16:02:47Z | +``` + +The full artifact contents: + +``` +rvs-nightly-report-/ +├── SUMMARY.md +└── rvs_level_4.log +``` + +Artifact retention is 30 days. + +## Log privacy (runner vs target) + +GitHub Actions always prints **Runner name** and **Machine name** in the **Set up job** +log group before any step from this repository runs. That text is emitted by the +[actions/runner](https://github.com/actions/runner) application when it accepts a job; +it is **not** produced by `rvs_nightly_test.sh` or this workflow YAML. + +**There is no supported way to suppress or hide those two lines** from repository +code (no workflow flag, secret, `::add-mask::`, or script can remove them — masking +only affects output *after* a step registers a value, and does not rewrite the setup +header GitHub has already recorded). + +What you *can* do on the infrastructure side: + +| Goal | Practical option | +|---|---| +| The literal strings should not identify a lab asset | Register the self-hosted runner under a **generic** name in GitHub (**Settings → Actions → Runners**), and/or use a hostname that is not inventory-sensitive — the lines will still exist, but the values are less revealing. | +| Avoid self-hosted runner logs entirely for orchestration | Run the `install-rvs-on-target` / `run-rvs-level-4` jobs on **GitHub-hosted** runners (`ubuntu-latest`) **only if** the target node is reachable from the public internet on SSH (unusual for internal lab hosts). | + +`validate-config` intentionally does **not** print the orchestrator hostname so +we do not duplicate host identity in our own script output. + +## GitHub runner vs target node + +The **install-rvs-on-target** and **run-rvs-level-4** jobs use +`${{ vars.RVS_NIGHTLY_INDEX_RUNNER_LABEL || vars.RVS_TEST_RUNNER_LABEL || 'self-hosted' }}` +and `${{ vars.RVS_TEST_RUNNER_LABEL || 'self-hosted' }}` respectively so **index resolution, tarball download, and `scp`** +happen only on the orchestrator. The runner only orchestrates — it doesn't need a GPU or ROCm. You can: + +- Reuse an existing self-hosted runner that already has SSH access to the lab **and** HTTPS egress to the tarball hosts. +- Use a small purpose-built orchestration runner (any Linux box with `ssh`/`scp`/`curl` and outbound HTTPS). +- **GitHub-hosted `ubuntu-latest`** is not used for tarball resolution or download; those steps must run where lab egress allows. + +The **target node** is what needs the GPU, ROCm, and `NOPASSWD` sudo. It does **not** need to be a registered GitHub runner. + +If the chosen orchestrator runner is busy with another job, this workflow's +`concurrency:` group (`rvs-nightly-${{ github.workflow }}`, +`cancel-in-progress: false`) will queue the run rather than cancel the +running one. + +## Verifying the pipeline end-to-end + +After committing the workflow file to `master`, the fastest sanity check +is: + +```bash +gh workflow run rvs-nightly-tests.yml +``` + +Watch the Actions tab for: + +1. **install-rvs-on-target** resolves a tarball URL (`Latest tarball : amdrocm-rvs-…`) on the orchestrator only. +2. **Validate configuration** (same job) prints `Target ROCm path`, `Remote work dir`, and `Expected RVS binary` (SSH target host/user are **not** printed). +3. **Setup SSH for target node** verifies SSH connectivity (no host identity printed to logs). +4. **Download tarball on orchestrator** then **Copy tarball to target** — the `.tar.gz` is never fetched on the GPU node. +5. **Verify ROCm prerequisites on target node** prints `::notice::ROCm prerequisites OK on target node at `. +6. **Install RVS on target node** prints the detected `ROCM_MAJOR`, the chosen `Target ROCm path`, and `Installed RVS at: /extras-/bin/rvs`. +7. **Verify RVS binary library resolution on target node** prints `::notice::RVS binary's library dependencies resolved OK on target, all from `. +8. **run-rvs-level-4** completes; the run summary shows the results table with the Level 4 row and overall PASS/FAIL. + +## Debugging a failed run + +| Symptom | Likely cause | +|---|---| +| **install-rvs-on-target** exits with "No tarball index URL…" | `RVS_TARBALL_INDEX_URL` is unset and `workflow_dispatch.tarball_url` was not provided. Set the secret to your HTTPS directory index (e.g. `https:///nightly/rvs/tar/`) or pass `tarball_url` for that run. | +| **Resolve latest tarball URL** fails with "Could not resolve a tarball URL" | The index page returned no matches. Confirm network access from the orchestrator to the index host, inspect the HTML listing for `amdrocm*-rvs-*-Linux.tar.gz` links, and/or set secret `RVS_TARBALL_INDEX_URL` to a valid listing URL. | +| **install-rvs-on-target** stuck "Queued" | No runner online for `vars.RVS_NIGHTLY_INDEX_RUNNER_LABEL` or `vars.RVS_TEST_RUNNER_LABEL` (see Actions → Runners). That job performs index `curl`, tarball download, and `scp` to the target. | +| **run-rvs-level-4** stuck "Queued" | No orchestrator runner online with the label in `vars.RVS_TEST_RUNNER_LABEL`. | +| **Validate target node configuration** fails: `No target node configured` | Neither `inputs.target_node` nor `secrets.RVS_TARGET_NODE` is set. Add the secret in Settings → Secrets and variables → Actions → Secrets, or pass `target_node` via `workflow_dispatch`. | +| **Setup SSH key for target node** fails: `secrets.RVS_TARGET_SSH_KEY is not set` | The required secret is missing. Add the private SSH key as a repo secret. | +| **Setup SSH key for target node** fails: `Permission denied (publickey)` | The key in `RVS_TARGET_SSH_KEY` isn't authorized on the target node for `$TARGET_USER`, or the key format is wrong. Verify from a workstation with `ssh -i -o BatchMode=yes @ true` (use your real user, host, and key; exit code 0 means the key works). | +| **Setup SSH key for target node** fails: `Connection timed out` / `Connection refused` | Network reachability problem between the orchestrator runner and the target node. Check firewall / VPN / bastion routing. | +| **Setup SSH key for target node** fails: `Host key verification failed` | `ssh-keyscan` couldn't pre-seed the key and `accept-new` rejected it (rare). Remove any stale entry for the target in the runner's `known_hosts`, or pre-populate it manually. | +| **Verify ROCm prerequisites on target node** fails: ` does not exist on the target node` | The configured `target_rocm_path` / `vars.RVS_TARGET_ROCM_PATH` doesn't point at a real directory on the target. Verify by `ssh @ 'ls -d '`. | +| **Verify ROCm prerequisites on target node** fails: `/bin/rocminfo: No such file or directory` | The install at `$TARGET_ROCM_PATH` is incomplete — missing `bin/rocminfo`. Either pick a different `target_rocm_path` (one that contains a complete ROCm distribution) or reinstall ROCm at the configured path. | +| **Verify ROCm prerequisites on target node** fails: `rocminfo` exits non-zero | Driver issue or runtime libs missing. Common causes: `amdgpu` kernel driver not loaded (`lsmod \| grep amdgpu`, then `sudo modprobe amdgpu` and check `dmesg`); `$TARGET_USER` not in `video`/`render` groups; `$TARGET_ROCM_PATH/lib` missing or broken. The `rocminfo` error message in the log usually pinpoints the cause. | +| **Verify ROCm prerequisites on target node** fails: `amd-smi version` exits non-zero | Either `amd-smi` is missing from `$TARGET_ROCM_PATH/bin/`, or the AMDSMI library under `$TARGET_ROCM_PATH/lib` is broken/missing. Reinstall or repoint `target_rocm_path` at a complete install. | +| **Verify RVS binary library resolution on target node** fails: `RVS binary has unresolved library dependencies on target` | One or more libraries showed up as `not found` in `ldd` output. The `LD_LIBRARY_PATH` (built from `$INSTALL_DIR/lib` + `$TARGET_ROCM_PATH/{lib,lib/llvm/lib,lib/rocm_sysdeps/lib}`) didn't have them, RUNPATH didn't have them, and `ldconfig` didn't have them either. Usually means `$TARGET_ROCM_PATH` is incomplete (missing libraries under `lib/`, `lib/llvm/lib/`, or `lib/rocm_sysdeps/lib/`). The step's `ldd` output above the error names the specific missing library. | +| **Install RVS on target node** fails: `Cannot parse ROCm major version from tarball name` | The tarball doesn't match `^amdrocm-`. Either pin a correctly-named tarball with `tarball_url`, or fix the upstream filename. | +| **Install RVS on target node** fails: `sudo: a password is required` | `$TARGET_USER` doesn't have `NOPASSWD` sudo on the target. Add a sudoers entry permitting `mkdir` and `tar` into `$TARGET_ROCM_PATH/extras-*` without a password. | +| **Install RVS on target node** fails: `rvs binary not found or not executable at /extras-/bin/rvs after install` | The tarball isn't rooted at `./bin/`, `./lib/`, etc. The extraction landed `bin/rvs` somewhere else inside `$TARGET_ROCM_PATH/extras-/`. Inspect the step log (`ls -la` output) to see the actual layout; the workflow may need `--strip-components=` added to the `tar` invocation. | +| **Verify RVS binary library resolution on target node** fails with `not found` | A ROCm runtime library is missing or not in `RPATH` on the target. The step prints the `ldd` output; install the matching ROCm component (typically `rocm-llvm`, `rocm-core`, `hip-runtime-amd`). | +| **Run RVS level 4 on target node** exits non-zero immediately (after both verify steps passed) | RVS plugin's own dependency missing on the target (e.g. `libpci3` on Debian). Check `rvs_level_4.log` in the artifact for the specific error. | +| **Collect logs from target node** warns: `No log files retrieved from target node` | The level steps exited so early they didn't produce any output, or `$REMOTE_WORK_DIR` was wiped. Inspect the level-step logs in the run UI for the original error. | +| **Create Test Report** / **Build test report**: `Required environment variable TARBALL_URL is not set` | The install job’s `tarball_url` output was empty when passed into the report job (often URLs with `%`, `&`, or other characters that break the old `echo "tarball_url=$URL"` → `GITHUB_OUTPUT` form). The workflow now writes that URL with delimiter syntax; upgrade to current `rvs-nightly-tests.yml`. The report step also tolerates a missing URL and still writes `SUMMARY.md` with the tarball filename. | +| Daily package build skipped or failed | Nightly tests only start after a **successful** **Build Relocatable Packages** run. If the scheduled package build failed or was skipped, nightly tests will not run until the next successful build (or trigger manually via `workflow_dispatch`). | + +## Retargeting at a different node + +There are three ways to point the workflow at a different node, in increasing order of permanence: + +1. **Single run, from the Actions UI:** Run workflow → fill in `target_node` and `target_rocm_path` (and optionally `target_user` / `remote_work_dir`). Anything left blank falls back to the matching repo configuration value — `target_node` and `target_user` read from the **secrets** `RVS_TARGET_NODE` / `RVS_TARGET_USER` (both masked as `***` in run logs); `target_rocm_path` / `remote_work_dir` read from repo Variables. `target_node` and `target_rocm_path` have no hard-coded defaults — the workflow fails fast if neither input nor stored value is set for either. `target_user` falls through to the orchestrator runner's local user; `remote_work_dir` falls through to `/tmp/rvs-nightly-`. +2. **Single run, from `gh` CLI:** + ```bash + gh workflow run rvs-nightly-tests.yml \ + -f target_node="" \ + -f target_rocm_path="" + ``` +3. **Permanent change:** update repo secrets `RVS_TARGET_NODE` / `RVS_TARGET_USER` (and optionally `RVS_TARBALL_INDEX_URL`) and repo variable `RVS_TARGET_ROCM_PATH` (and optionally repo variable `RVS_REMOTE_WORK_DIR`). All subsequent post-build (`workflow_run`) and manual runs will pick these up unless an input overrides them. + +The same `RVS_TARGET_SSH_KEY` secret is reused across nodes — make sure the +public counterpart of that key is added to `$TARGET_USER`'s `~/.ssh/authorized_keys` +on **every** node you intend to point the workflow at. + +## References + +- [RVS source](../../README.md) +- [`build-relocatable-packages.yml`](./build-relocatable-packages.yml) and [`README_BUILD_PACKAGES.md`](./README_BUILD_PACKAGES.md) — the upstream packaging pipeline that produces these tarballs +- [GitHub Actions: scheduled events](https://docs.github.com/en/actions/using-workflows/events-that-trigger-workflows#schedule) +- [GitHub Actions: encrypted secrets](https://docs.github.com/en/actions/security-guides/using-secrets-in-github-actions) — for `RVS_TARGET_SSH_KEY`, `RVS_TARBALL_INDEX_URL`, etc. diff --git a/.github/workflows/README_PR_TESTS.md b/.github/workflows/README_PR_TESTS.md new file mode 100644 index 000000000..e7c0b1abe --- /dev/null +++ b/.github/workflows/README_PR_TESTS.md @@ -0,0 +1,75 @@ +# RVS PR Tests Workflow + +This document describes [`.github/workflows/rvs-pr-tests.yml`](./rvs-pr-tests.yml). It mirrors [RVS Nightly Tests](./README_NIGHTLY_TESTS.md) **orchestration** (self-hosted runner resolves the artifact, `curl` + `scp` to the GPU target, SSH-driven install, `rvs -r 4`, Markdown report) but targets the **Linux relocatable tarball** (`amdrocm*-rvs-*-Linux.tar.gz`) — not the Ubuntu `.deb` by default. A **direct `.deb` URL** is still supported for manual runs. + +## Triggers + +| Trigger | When | Behavior | +|---|---|---| +| `workflow_run` | After **[Build Relocatable Packages](https://github.com/ROCm/ROCmValidationSuite/actions/workflows/build-relocatable-packages.yml)** completes | Runs **only** when the triggering run **`conclusion` is `success`** and the triggering event is **`pull_request`**. Resolves the tarball at `…/rvs//merge//manylinux_2_28/` (same path layout as [`.github/scripts/rvs-s3-upload-route.sh`](../../.github/scripts/rvs-s3-upload-route.sh) for PR uploads). | +| `workflow_dispatch` | Manual | Always runs (subject to secrets/inputs). | + +There is **no** `schedule` and **no** CDN polling or Actions cache marker. + +Install logic is in [`rvs_pr_test.sh`](../../rvs_pr_test.sh): **`.tar.gz`** → `tar -xzf` under `/opt/rocm/extras-/` (same as nightly); **`.deb`** → `dpkg` path. + +## How the package URL is chosen (`resolve-package-url`) + +**After a successful PR build (`workflow_run`)** + +1. Read **`github.event.workflow_run.pull_requests[0].number`** and **`github.event.workflow_run.run_number`** (the Build workflow’s run number, same as **`GITHUB_RUN_NUMBER`** during S3 upload in **Build Relocatable Packages**). +2. Set secret **`RVS_PR_DEB_PACKAGE_URL`** (or **`RVS_PR_PACKAGE_URL`**) to the HTTPS **`…/rvs`** (or **`…/rvs/`**) base only — no `merge/` or `manylinux` segments (for example the CloudFront root you use for PR artifacts). +3. The workflow probes **`manylinux_2_28`** listings in this order (first that returns a listing wins): + - **`{base}/{pr}/merge/{run_number}/manylinux_2_28/`** — matches [`.github/scripts/rvs-s3-upload-route.sh`](../../.github/scripts/rvs-s3-upload-route.sh) PR layout (`rvs/${GITHUB_REF_NAME}/${GITHUB_RUN_NUMBER}/manylinux_2_28` with `GITHUB_REF_NAME` = `/merge`). + - **`{base}/{run_number}/merge/{pr}/manylinux_2_28/`** — alternate mirror layout. +4. [`rvs_pr_test.sh resolve-package-url`](../../rvs_pr_test.sh) treats that URL as a **`manylinux_*`** directory listing, selects the latest **`amdrocm*-rvs-*-Linux.tar.gz`**, then download → target install → **`rvs -r 4`** on the GPU host (see nightly README for SSH / `RVS_TARGET_*`). + +If **`pull_requests`** is empty (common for some fork PR flows), the job fails with a clear error — set up a manual run with `package_url` instead, or adjust branch/secret policy so the payload includes the PR. + +**Manual (`workflow_dispatch`)** + +- **`package_url`** or **`RVS_PR_DEB_PACKAGE_URL`** / **`RVS_PR_PACKAGE_URL`** may be: + - **`…/rvs/`** index root → crawler picks latest `rvs//merge//…/*.tar.gz` (legacy layout), + - a **`…/manylinux_*/`** directory listing → latest tarball in that folder, + - a **direct** `*-Linux.tar.gz` or **`.deb`** URL. + +## Repository configuration + +Use the same **secrets** and **variables** as nightly for SSH and ROCm (`RVS_TARGET_NODE`, `RVS_TARGET_USER`, `RVS_TARGET_SSH_KEY`, `RVS_TARGET_ROCM_PATH`, runner labels, etc.). See [README_NIGHTLY_TESTS.md](./README_NIGHTLY_TESTS.md#repository-configuration). + +**PR package URL secrets** + +| Name | Kind | Purpose | +|---|---|---| +| `RVS_PR_DEB_PACKAGE_URL` | Secret | **Required** for `workflow_run`: HTTPS base **`…/rvs`** or **`…/rvs/`** only (no `merge/` / `manylinux` path). **Manual runs:** same base, or `…/manylinux_*/` listing, or direct tarball / `.deb`. | +| `RVS_PR_PACKAGE_URL` | Secret | Legacy alias if `RVS_PR_DEB_PACKAGE_URL` is unset. | + +## Manual run + +```bash +gh workflow run rvs-pr-tests.yml +``` + +Examples — set your real values in secrets or `-f package_url`: + +**`…/rvs/` index root** (crawler uses `rvs//merge//…`): + +```bash +gh workflow run rvs-pr-tests.yml \ + -f package_url="https:///rvs/" \ + -f target_node="" +``` + +**Direct tarball URL**: + +```bash +gh workflow run rvs-pr-tests.yml \ + -f package_url="https:////amdrocm7-rvs-…-Linux.tar.gz" \ + -f target_node="" +``` + +Artifacts: `rvs-pr-logs-`, `rvs-pr-report-`. + +## Target prerequisites + +Same as nightly: for **tarball** installs, target layout matches **relocatable RVS** under `/opt/rocm/extras-/`; **`sudo -n`** if `/opt` is not user-writable. For **`.deb`**, passwordless **`sudo`** for `dpkg` / `apt-get`. ROCm at **`RVS_TARGET_ROCM_PATH`**. Orchestrator needs HTTPS egress to your CDN and SSH/SCP to the target. diff --git a/.github/workflows/README_RELEASE_TESTS.md b/.github/workflows/README_RELEASE_TESTS.md new file mode 100644 index 000000000..a1663c268 --- /dev/null +++ b/.github/workflows/README_RELEASE_TESTS.md @@ -0,0 +1,205 @@ +# RVS Release Tests Workflow + +This document describes [`.github/workflows/rvs-release-tests.yml`](./rvs-release-tests.yml), +which picks up the **latest RVS release tarball** from the release tarball index +(`secrets.RVS_RELEASE_TARBALL_INDEX_URL`, passed through workflow `env.TARBALL_INDEX_URL` — +**no default URL in the repo**), copies it to a +**dedicated release test GPU node** over SSH, installs it there, and runs **RVS level 5** +(full stress) on that node. + +Release testing is intentionally separate from [nightly tests](./README_NIGHTLY_TESTS.md): +different tarball channel, different target-node secrets, and a heavier test level. + +The GitHub Actions runner is only an **orchestrator** — it does not need a GPU or ROCm +locally. Install, binary verification, and `rvs -r 5` run on the remote target. + +## What it does + +``` +workflow_run (release/* build) / workflow_dispatch + │ + ▼ +gate-release-trigger [utility runner] — skip non-release workflow_run events + ▼ +install-rvs-on-target [self-hosted orchestrator] ──ssh──▶ [release GPU node] + │ resolve release index + validate paths (orchestrator only) + │ setup-ssh → download .tar.gz on runner → scp to target + │ verify-rocm → install-rvs → verify-rvs-binary + ▼ +run-rvs-level-5 [self-hosted orchestrator] ──ssh──▶ [release GPU node] + │ rvs_release_test.sh: run-level5 (release tarballs only), collect-logs, capture-versions + │ upload intermediate logs artifact; cleanup remote work dir + ▼ +create-test-report [utility runner] + │ rvs_release_test.sh build-report → SUMMARY.md + final artifact + ▼ +artifact: rvs-release-report- +``` + +Install, test, and report logic lives in [`rvs_release_test.sh`](../../rvs_release_test.sh) +at the repo root — independent from [`rvs_nightly_test.sh`](../../rvs_nightly_test.sh) (nightly) +and [`rvs_pr_test.sh`](../../rvs_pr_test.sh) (PR). + +## Triggers + +| Trigger | When it runs | +|---|---| +| `workflow_run` | After **Build Relocatable Packages** completes **successfully** on a `release/*` branch. Builds on `main`, feature branches, or failed release builds do **not** start this workflow. | +| `workflow_dispatch` | Manual run from the Actions UI (or `gh workflow run`). Always eligible; does not require a prior package build. | + +There is no `schedule` trigger — release validation is tied to release package publication +or explicit manual runs. + +## Manual dispatch inputs + +| Input | Default | Description | +|---|---|---| +| `tarball_url` | _(empty → latest from release index)_ | Download this exact URL instead of scraping the release index. | +| `target_node` | _(empty → `secrets.RVS_RELEASE_TARGET_NODE`)_ | Hostname or IP of the release test GPU node. | +| `target_user` | _(empty → `secrets.RVS_RELEASE_TARGET_USER`)_ | SSH user on the target node. | +| `remote_work_dir` | _(empty → `vars.RVS_RELEASE_REMOTE_WORK_DIR`, then `/tmp/rvs-release-`)_ | Staging and log directory on the target; removed at job end. | +| `target_rocm_path` | _(empty → `secrets.RVS_RELEASE_TARGET_ROCM_PATH`)_ | Absolute path to the ROCm install root on the target. | + +Workflow inputs override repo secrets for that run only. + +Examples: + +```bash +# Latest release tarball from the index secret, default release target secrets +gh workflow run rvs-release-tests.yml + +# Pin a specific release tarball +gh workflow run rvs-release-tests.yml \ + -f tarball_url="/amdrocm7-rvs-1.4.21-r0711.20260623-Linux.tar.gz" + +# One-off run against a different node (secrets unchanged) +gh workflow run rvs-release-tests.yml \ + -f target_node="" \ + -f target_user="" \ + -f target_rocm_path="" +``` + +## Repository configuration + +### Secrets (release-specific) + +All release target settings are **secrets** (masked as `***` in logs): + +| Name | Required? | Purpose | +|---|---|---| +| `RVS_RELEASE_TARGET_NODE` | **Required** *(unless every run sets `target_node` input)* | Hostname or IP of the release validation GPU node. | +| `RVS_RELEASE_TARGET_USER` | optional | SSH user on the release target. If unset, SSH falls back to the orchestrator runner's local user. | +| `RVS_RELEASE_TARGET_SSH_KEY` | **Required** | Private SSH key authorized on the release target for `RVS_RELEASE_TARGET_USER`. | +| `RVS_RELEASE_TARGET_ROCM_PATH` | **Required** *(unless every run sets `target_rocm_path` input)* | Absolute path to the ROCm install root on the release target (`bin/rocminfo`, `bin/amd-smi`, `lib/`, etc.). | +| `RVS_RELEASE_TARBALL_INDEX_URL` | **Required** *(unless every run provides `workflow_dispatch` `tarball_url`; `workflow_run` needs this secret)* | HTTPS directory listing URL scraped for the latest `amdrocm*-rvs-*-Linux.tar.gz` release tarball. Copied into workflow `env.TARBALL_INDEX_URL`. GitHub masks secret values in logs when printed. | + +These are **independent** from nightly/PR secrets (`RVS_TARGET_*`, `RVS_TARBALL_INDEX_URL`). Use a dedicated +release-validation host so full-stress level 5 does not contend with nightly level 4. + +### Variables (optional) + +| Name | Default | Purpose | +|---|---|---| +| `RVS_RELEASE_INDEX_RUNNER_LABEL` | falls back to `RVS_NIGHTLY_INDEX_RUNNER_LABEL`, then `RVS_TEST_RUNNER_LABEL`, then `self-hosted` | Runner label for **install-rvs-on-target** (index resolve + download + scp). | +| `RVS_RELEASE_TEST_RUNNER_LABEL` | falls back to `RVS_TEST_RUNNER_LABEL`, then `self-hosted` | Runner label for **run-rvs-level-5**. | +| `RVS_RELEASE_REMOTE_WORK_DIR` | `/tmp/rvs-release-` | Fixed remote work dir when not using the per-run input. | + +## How the latest release tarball is picked + +Unless `workflow_dispatch.tarball_url` is set, **install-rvs-on-target** scrapes +`secrets.RVS_RELEASE_TARBALL_INDEX_URL` (workflow `env.TARBALL_INDEX_URL`): + +```bash +curl -sL "$INDEX_URL" \ + | grep -oE 'amdrocm[0-9]*-rvs-[0-9A-Za-z._\-]+-Linux\.tar\.gz' \ + | sort -uV \ + | tail -n 1 +``` + +The version-sorted newest filename is downloaded on the orchestrator and `scp`'d to the +target. The target never fetches the index or tarball URL directly. + +Release tarballs are published by +[Build Relocatable Packages](./build-relocatable-packages.yml) when a `release/*` +branch build completes (CentOS job uploads the `.tar.gz` to the release tarball prefix). + +## The test — level 5 (full stress) + +On the target node (with `$RVS_BIN` = `/opt/rocm/extras-/bin/rvs`): + +```bash +"$RVS_BIN" -r 5 +``` + +Level 5 runs the full stress suite (extended duration vs level 4). Expect **hours** of +runtime depending on GPU SKU and configuration. The **run-rvs-level-5** job timeout is +**480 minutes** (8 hours); increase `timeout-minutes` in the workflow if your hardware +needs longer. + +Stdout/stderr is captured to `$REMOTE_WORK_DIR/reports/rvs_level_5.log` on the target, +then copied back to `./reports/` on the orchestrator. + +See [RVS user guide — test levels](../../docs/ug1main.md) for what level 5 includes. + +## Test report + +`create-test-report` writes `reports/SUMMARY.md` and uploads artifact +`rvs-release-report-`: + +```markdown +# RVS Release Test Report + +| Field | Value | +|---|---| +| Tarball | `amdrocm7-rvs-…-Linux.tar.gz` | +| Overall result | **PASS** | + +## Results + +| Test | Command | Result | Exit | +|---------|--------------------------------------|:------:|-----:| +| Level 5 | `/opt/rocm/extras-7/bin/rvs -r 5` | PASS | 0 | +``` + +Artifact contents: + +``` +rvs-release-report-/ +├── SUMMARY.md +└── rvs_level_5.log +``` + +## Target node prerequisites + +Same requirements as nightly tests — see +[README_NIGHTLY_TESTS.md — prerequisites](./README_NIGHTLY_TESTS.md#target-node-prerequisites) +for SSH, `NOPASSWD` sudo, ROCm layout, and the manual validation one-liner. + +Quick checklist for the **release** node: + +- SSH reachable from the orchestrator; `RVS_RELEASE_TARGET_SSH_KEY` authorized. +- ROCm at `RVS_RELEASE_TARGET_ROCM_PATH` with `rocminfo` and `amd-smi` working. +- GPU(s) suitable for sustained level 5 stress (power/thermal headroom, no conflicting jobs). +- `sudo -n` if `/opt/rocm/extras-` is not user-writable. + +## Troubleshooting + +| Symptom | Likely cause | +|---|---| +| Workflow skipped after a main-branch package build | Expected — only `release/*` builds pass `gate-release-trigger`. | +| `No tarball index URL` | Set `RVS_RELEASE_TARBALL_INDEX_URL` secret or pass `tarball_url` on manual dispatch. | +| `Required environment variable TARGET_NODE is not set` | Add `RVS_RELEASE_TARGET_NODE` secret or `target_node` input. | +| Level 5 job cancelled at 8h | Increase `timeout-minutes` on **run-rvs-level-5** for slower SKUs. | +| `ldd` / library errors before level 5 | Fix ROCm or RVS install paths on the target; see nightly README verify steps. | + +For orchestrator SSH, index resolution, and log collection details, see +[README_NIGHTLY_TESTS.md](./README_NIGHTLY_TESTS.md). + +## Related workflows + +| Workflow | Script | Tarball source | Test level | Target secrets | +|---|---|---|---|---| +| [rvs-nightly-tests.yml](./rvs-nightly-tests.yml) | `rvs_nightly_test.sh` | `RVS_TARBALL_INDEX_URL` | 4 | `RVS_TARGET_*` | +| [rvs-pr-tests.yml](./rvs-pr-tests.yml) | `rvs_pr_test.sh` | PR package URL | 4 | `RVS_TARGET_*` | +| **rvs-release-tests.yml** | **`rvs_release_test.sh`** | **`RVS_RELEASE_TARBALL_INDEX_URL`** | **5** | **`RVS_RELEASE_TARGET_*`** | +| [build-relocatable-packages.yml](./build-relocatable-packages.yml) | — | _(builds packages)_ | — | — | diff --git a/.github/workflows/README_UNSIGNED_RC_PROMOTION.md b/.github/workflows/README_UNSIGNED_RC_PROMOTION.md new file mode 100644 index 000000000..f4460bee9 --- /dev/null +++ b/.github/workflows/README_UNSIGNED_RC_PROMOTION.md @@ -0,0 +1,303 @@ +# Unsigned Release Candidate Promotion + +This document describes [`.github/workflows/unsigned-release-candidate-promotion.yml`](./unsigned-release-candidate-promotion.yml), which copies RVS packages for a specific build number from the **signed release source** (`release/rvs/`) into the **unsigned staging area** (`release/unsigned/`) and regenerates the APT/YUM repository metadata there. The resulting layout uses the same `packages/deb/`, `packages/rpm/x86_64/`, and `tarball/` sub-structure as `nightly/unsigned/` so that the same signing CI can consume either path without changes. + +## Purpose + +When a release branch build completes, its packages land under `release/rvs/{deb,rpm,tar}/`. Before those packages can be signed and published to consumers, they must be promoted into `release/unsigned/` — the staging prefix that signing CI monitors. This workflow performs that promotion on demand, filtered by the build number encoded in each package filename. Packages are placed under `release/unsigned/packages/{deb,rpm}/` and `release/unsigned/tarball/`. + +## Trigger + +The workflow runs only on **manual dispatch** (`workflow_dispatch`). There is no scheduled or automatic trigger — promotion is always an explicit human action. + +### Input + +| Input | Required | Description | +|-------|----------|-------------| +| `run_number` | **Yes** | GitHub Actions run number of the `build-relocatable-packages` workflow run that produced the release packages (e.g. `12345`). Release packages are named with this number as their release segment: `amdrocm7-rvs_1.3.15-12345_amd64.deb`, `amdrocm7-rvs-1.3.15-12345..x86_64.rpm`, `amdrocm7-rvs-1.3.15-12345-Linux.tar.gz`. | + +**Matching uses format-specific delimiters, not a plain substring.** Each format step looks for the run number bracketed by the characters that surround it in the filename: + +| Format | Pattern used | Example filename | +|--------|-------------|-----------------| +| DEB | `*"-_"*` | `amdrocm7-rvs_1.3.15-12345_amd64.deb` | +| RPM | `*"-."*` | `amdrocm10-rvs-1.6.131-12345..x86_64.rpm` | +| TAR | `*"--Linux"*` | `amdrocm7-rvs-1.3.15-12345-Linux.tar.gz` | + +This prevents run number `123` from false-matching a file built by run `1234`. Exactly one file per format must match; the step fails on zero or more than one match. + +## S3 layout + +``` +s3:/// +├── release/rvs/ ← source (written by build-relocatable-packages.yml) +│ ├── deb/ +│ │ └── amdrocm*-rvs*.deb +│ ├── rpm/ +│ │ └── amdrocm*-rvs*.rpm +│ └── tar/ +│ └── amdrocm*-rvs*.tar.gz +│ +└── release/unsigned/ ← destination (written by this workflow) + ├── packages/ + │ ├── deb/ + │ │ ├── conf/ # reprepro state (internal; not for apt clients) + │ │ ├── pool/main/…/amdrocm*-rvs*.deb + │ │ └── dists/stable/ + │ │ ├── Release + │ │ └── main/binary-amd64/Packages(.gz) + │ └── rpm/ + │ └── x86_64/ + │ ├── amdrocm*-rvs*.rpm + │ └── repodata/ + │ ├── repomd.xml + │ ├── primary.xml.gz + │ ├── filelists.xml.gz + │ └── other.xml.gz + ├── tarball/ + │ ├── amdrocm*-rvs*.tar.gz + │ └── amdrocm*-rvs*.tar.gz.sha256 + └── latest.json # signing CI entry point +``` + +The source paths (`release/rvs/`) are populated by `build-relocatable-packages.yml` when a `release/**` branch is pushed or manually dispatched. This workflow reads from those paths and writes to `release/unsigned/`. + +## What the workflow does + +The single job (`promote-release-unsigned`) runs these steps in order: + +### 1. Install AWS CLI + +Same pattern as `build-relocatable-packages.yml`: uses `pip install awscli` with a `--break-system-packages` fallback so it works on both Ubuntu 22.04 and Ubuntu 24.04 hosted runners. + +### 2. Configure AWS credentials + +Uses OIDC (`assume-role-with-web-identity`) with `secrets.AWS_ROLE_ARN`. No long-term access keys are stored. Credentials are masked in logs. + +### 3. Validate inputs + +Fails fast if `run_number` or `AWS_S3_BUCKET` are empty. Prints the source and destination S3 prefixes to the log. + +### 4. Install packaging tools + +Installs `reprepro`, `dpkg-dev`, and `createrepo-c` (or `createrepo`) via `apt-get`. These are needed to maintain the APT archive structure and RPM repodata. + +### 5. Copy DEB packages and rebuild APT archive + +- Downloads the existing `release/unsigned/packages/deb/{conf,pool,dists}/` from S3 into a staging directory (**accumulate mode** — no `--delete`) +- Bootstraps a `reprepro` distributions config (`Suite: stable`, `Codename: stable`) if the archive does not yet exist +- Lists all keys under `release/rvs/deb/` and downloads those matching `amdrocm*-rvs*.deb` and containing `build_number` +- For each matching `.deb`, runs `reprepro includedeb stable` — idempotently removing the same Package+Version from the local archive first if it is already present, so S3 `PutObject` can overwrite the pool object without requiring `s3:DeleteObject` +- Syncs `conf/`, `pool/`, and `dists/` back to S3 (no `--delete`) + +### 6. Copy RPM packages and rebuild YUM repodata + +- Downloads existing `release/unsigned/packages/rpm/x86_64/` RPM files from S3, excluding `repodata/` (which will be fully regenerated). `aws s3 sync` exits 0 for a nonexistent prefix (first promotion), so no error-suppression is needed. +- Lists `release/rvs/rpm/` by writing to a temp file (not a process substitution) so a non-zero exit from the `aws` CLI propagates under `set -euo pipefail` +- Downloads the one matching `.rpm` into a local `x86_64/` subdirectory — fails if zero or more than one file matches +- Computes the SHA-256 of the downloaded RPM and saves it as a step output (`rpm_fname`, `rpm_sha256`) +- Runs `createrepo_c --simple-md-filenames --no-database --compress-type gz` (fallback: `createrepo`) on the `x86_64/` directory, producing `repomd.xml`, `primary.xml.gz`, `filelists.xml.gz`, and `other.xml.gz` (same names as [stable extras repodata](https://stable.repo.amd.com/rocm/extras/rvs/packages/rhel8/x86_64/repodata/)). `repomd.xml.asc` is added later by the signing job. +- Syncs `x86_64/` back to `s3:///release/unsigned/packages/rpm/x86_64/` + +RPMs are placed under `x86_64/` so that the signing CI and yum/dnf clients can use `release/unsigned/packages/rpm/x86_64/` as the `baseurl` directly. + +### 7. Copy TAR packages + +- Lists `release/rvs/tar/` by writing to a temp file so aws errors propagate +- Downloads the one matching `.tar.gz` and its `.sha256` sidecar if present — fails if zero or more than one tarball matches +- Generates a SHA-256 sidecar if the source did not include one. `sha256sum` is run with `cd "${STAGING}" && sha256sum "${basename}"` so the path recorded in the sidecar file is the bare filename (e.g. `abc123 amdrocm7-rvs-….tar.gz`), not the runner temp path (`/tmp/…/amdrocm7-rvs-….tar.gz`). This matches what `sha256sum -c` expects on the signing host. +- Saves the SHA-256 as a step output (`tar_fname`, `tar_sha256`) +- Copies both `.tar.gz` and `.tar.gz.sha256` to `release/unsigned/tarball/` + +### 8. Publish `release/unsigned/latest.json` + +Assembles and uploads `release/unsigned/latest.json`. All three formats are required — the step fails immediately if any step output from the promote steps is missing. No partial writes: either all three are present or the file is not written. + +- Fetches `release/unsigned/packages/deb/dists/stable/main/binary-amd64/Packages` (hard failure if absent) and parses it to find DEB pool paths matching `run_number` +- Takes RPM filename and SHA-256 directly from the `promote-rpm` step output — no re-download +- Takes TAR filename and SHA-256 directly from the `promote-tar` step output — no re-download +- Writes and uploads `release/unsigned/latest.json`, matching the schema that `rvs-unsigned-publish-latest.sh` validates (`rpm.sha256`, `tar.sha256`, `tar.sha256_sidecar_key` are all present) + +The `latest.json` schema: + +```json +{ + "github_run_id": "12345678", + "github_sha": "abc123...", + "rocm_version": null, + "run_number": "12345", + "published_at": "2026-04-23T12:34:56Z", + "deb": { + "github_run_id": "12345678", + "deb_prefix": "release/unsigned/packages/deb", + "packages": [ + { + "filename": "amdrocm7-rvs_1.3.15-12345_amd64.deb", + "pool_key": "pool/main/a/amdrocm7-rvs/amdrocm7-rvs_1.3.15-12345_amd64.deb", + "s3_key": "release/unsigned/packages/deb/pool/main/a/amdrocm7-rvs/amdrocm7-rvs_1.3.15-12345_amd64.deb" + } + ] + }, + "rpm": { + "filename": "amdrocm7-rvs-1.3.15-12345..x86_64.rpm", + "s3_key": "release/unsigned/packages/rpm/x86_64/amdrocm7-rvs-1.3.15-12345..x86_64.rpm", + "sha256": "abc123def456..." + }, + "tar": { + "filename": "amdrocm7-rvs-1.3.15-12345-Linux.tar.gz", + "s3_key": "release/unsigned/tarball/amdrocm7-rvs-1.3.15-12345-Linux.tar.gz", + "sha256": "789abc012def...", + "sha256_sidecar_key": "release/unsigned/tarball/amdrocm7-rvs-1.3.15-12345-Linux.tar.gz.sha256" + } +} +``` + +### 9. Write job summary + +Writes a Markdown table to the GitHub Actions run summary listing the build number, count of promoted packages per format, and the source/destination S3 paths. + +## Accumulation semantics + +All S3 writes use **accumulate mode** (no `--delete`). Packages from previous promotions remain in `release/unsigned/`. The `reprepro` DEB archive accumulates historically: older `.deb` versions remain in the pool (the reprepro db tracks each Package+Version). RPM repodata is fully regenerated from all RPMs present in the staging directory after the new ones are added. + +Signing CI should always read `release/unsigned/latest.json` to find the exact `s3_key` values for a specific promotion — not list the prefix directly. + +## All three formats are required + +All of DEB, RPM, and TAR must have exactly one matching package. Any of the following causes the relevant step to fail hard (non-zero exit): + +- Zero files match the `run_number` for that format (the build may not have uploaded to `release/rvs/`, or the wrong run number was entered) +- More than one file matches (should not normally occur since each GitHub run number is unique, but would indicate duplicate files in the bucket) + +`latest.json` is never written with a subset of formats. It is only published when all three promote steps succeed. + +## Required configuration + +**Repository secret** (Settings → Secrets and variables → Actions → Secrets): + +| Secret | Purpose | +|--------|---------| +| `AWS_ROLE_ARN` | IAM role ARN to assume for S3 access via OIDC. The role must have `s3:PutObject`, `s3:GetObject`, and `s3:ListBucket` on `release/unsigned/*` and `release/rvs/*`. `s3:DeleteObject` is **not** required. | + +**Repository variable** (Settings → Secrets and variables → Actions → Variables): + +| Variable | Default | Purpose | +|----------|---------|---------| +| `AWS_S3_BUCKET` | _(required)_ | S3 bucket name. | +| `RUNNER_LABEL_UTILITY` | `ubuntu-latest` | Runner label for the promotion job. | + +**AWS IAM trust policy:** The role in `AWS_ROLE_ARN` must allow GitHub OIDC (`token.actions.githubusercontent.com`, audience `sts.amazonaws.com`) to assume it for this repository. + +## Relationship to nightly/unsigned + +| Attribute | `nightly/unsigned/` | `release/unsigned/` | +|-----------|--------------------|--------------------| +| Populated by | `build-relocatable-packages.yml` (scheduled default branch) | This workflow (manual dispatch) | +| Source packages | `nightly/rvs/` | `release/rvs/` | +| DEB path | `packages/deb/` | `packages/deb/` | +| DEB APT suite | `stable main` | `stable main` | +| RPM layout | packages in `packages/rpm/x86_64/`; repodata inside `packages/rpm/x86_64/repodata/` | packages in `packages/rpm/x86_64/`; repodata inside `packages/rpm/x86_64/repodata/` | +| RPM repodata | `createrepo_c --simple-md-filenames --no-database --compress-type gz` (`repomd.xml`, `primary.xml.gz`, `filelists.xml.gz`, `other.xml.gz`; `repomd.xml.asc` from signing) | Same flags, run on `x86_64/` | +| TAR path | `tarball/` | `tarball/` | +| `.tar.gz.sha256` sidecars | Always generated by build job | Generated here if absent in source | +| `latest.json` | Per-run, always overwrites | Per-promotion, always overwrites | +| Accumulation | Yes (no `--delete`) | Yes (no `--delete`) | +| Per-run fragments (`runs//`) | Yes (`deb.json`, `rpm-tar.json`) | No (not needed; promotion is explicit) | + +## Triggering the workflow + +**From the GitHub Actions UI:** + +1. Go to **Actions** → **Unsigned Release Candidate Promotion** +2. Click **Run workflow** +3. Enter the `run_number` — the run number of the `build-relocatable-packages` run that produced the release packages (e.g. `12345`) +4. Click **Run workflow** + +**From the `gh` CLI:** + +```bash +gh workflow run unsigned-release-candidate-promotion.yml \ + -f run_number="12345" +``` + +The `run_number` is the GitHub Actions run number shown on the `build-relocatable-packages` workflow run page (the integer in the URL and the run number column in the Actions tab). Because each run number is unique within the repository, it unambiguously identifies exactly one set of release packages. + +## Signing CI handoff + +After a successful promotion, signing CI reads `s3:///release/unsigned/latest.json`. The file lists: + +- `deb.packages[].s3_key` — pool paths for each `.deb` in the APT archive +- `rpm.s3_key` — path for the `.rpm` +- `tar.s3_key` and `tar.sha256_sidecar_key` — paths for the tarball and its SHA-256 sidecar + +Signing CI downloads the packages from those keys, signs them, and promotes the signed artifacts to the consumer-facing release repository. + +## Verifying the promotion + +After the workflow completes, check the S3 paths via the AWS CLI or console: + +```bash +BUCKET="" +BUILD_NUM="r0711.20260423" + +# DEB: verify APT index contains the package +aws s3 cp "s3://${BUCKET}/release/unsigned/packages/deb/dists/stable/main/binary-amd64/Packages" - \ + | grep -A5 "${BUILD_NUM}" + +# RPM: verify package and repodata (packages live under x86_64/) +aws s3 ls "s3://${BUCKET}/release/unsigned/packages/rpm/x86_64/" | grep "${BUILD_NUM}" +aws s3 ls "s3://${BUCKET}/release/unsigned/packages/rpm/x86_64/repodata/" + +# TAR: verify tarball and sidecar +aws s3 ls "s3://${BUCKET}/release/unsigned/tarball/" | grep "${BUILD_NUM}" + +# latest.json +aws s3 cp "s3://${BUCKET}/release/unsigned/latest.json" - +``` + +**Using the promoted repo with apt (unsigned staging, internal testing):** + +```bash +echo "deb [trusted=yes arch=amd64] https://.s3.amazonaws.com/release/unsigned/packages/deb/ stable main" \ + | sudo tee /etc/apt/sources.list.d/rvs-unsigned-release.list +sudo apt update +sudo apt install amdrocm7-rvs +``` + +**Using the promoted repo with yum/dnf (unsigned staging, internal testing):** + +```bash +cat <<'EOF' | sudo tee /etc/yum.repos.d/rvs-unsigned-release.repo +[rvs-unsigned-release] +name=RVS Unsigned Release Candidate RPM +baseurl=https://.s3.amazonaws.com/release/unsigned/packages/rpm/x86_64/ +enabled=1 +gpgcheck=0 +EOF +sudo yum install amdrocm7-rvs +``` + +> **Note:** `[trusted=yes]` (apt) and `gpgcheck=0` (yum) disable GPG verification. These repo definitions are for **internal staging and testing only**, before signing CI produces signed packages for public consumption. + +## Troubleshooting + +| Symptom | Likely cause | +|---------|-------------| +| `AWS_S3_BUCKET repository variable is not set` | The `AWS_S3_BUCKET` Actions variable is missing. Add it in Settings → Secrets and variables → Actions → Variables. | +| `Credentials could not be loaded` | The OIDC trust policy for `AWS_ROLE_ARN` does not cover this repository, or `AWS_ROLE_ARN` is not set as a repository secret. | +| `No .deb files for run number ''` | No DEB in `release/rvs/deb/` has `-_` in its filename. Confirm that the `build-relocatable-packages` run with that number ran against a `release/**` branch and uploaded packages. Verify with `aws s3 ls s3:///release/rvs/deb/ --recursive`. | +| `No .rpm files for run number ''` | Same for RPMs. Check `release/rvs/rpm/`. | +| `No .tar.gz files for run number ''` | Same for tarballs. Check `release/rvs/tar/`. | +| `N .rpm files match run number ''; expected exactly one` | Duplicate files in the bucket share the same run number. Inspect `release/rvs/rpm/` directly to identify and remove the duplicate. | +| `promote-rpm step output rpm_fname is missing` | The `promote-rpm` step either did not run or failed before writing its outputs. Check that step's logs. | +| `Packages index not found` in latest.json step | The DEB promote step succeeded (uploaded packages) but the Packages index was not found at `dists/stable/main/binary-amd64/Packages`. This indicates a reprepro or S3 sync failure in the DEB step. | +| `reprepro` fails on `includedeb` | The `.deb` control fields may have unexpected characters, or the `conf/distributions` file is corrupted. Delete `s3:///release/unsigned/packages/deb/conf/` to force a fresh archive on next run. | +| RPM repodata not updated | `createrepo_c` may not be available on the runner. The step falls back to `createrepo`; if neither is found, the step fails. The runner label in `RUNNER_LABEL_UTILITY` must resolve to a runner where at least one of those tools can be installed via `apt-get`. | + +## References + +- [`build-relocatable-packages.yml`](./build-relocatable-packages.yml) — upstream pipeline that produces packages in `release/rvs/` +- [`README_BUILD_PACKAGES.md`](./README_BUILD_PACKAGES.md) — detailed documentation for the build pipeline, including the nightly/unsigned layout it mirrors +- [`rvs-deb-unsigned-repo.sh`](../scripts/rvs-deb-unsigned-repo.sh) — reprepro accumulate logic used by the nightly unsigned pipeline (same pattern as the DEB step here) +- [`rvs-unsigned-publish-latest.sh`](../scripts/rvs-unsigned-publish-latest.sh) — latest.json publisher for nightly/unsigned (same schema as `release/unsigned/latest.json`) +- [`rvs-s3-upload-route.sh`](../scripts/rvs-s3-upload-route.sh) — S3 path routing for build jobs diff --git a/.github/workflows/build-relocatable-packages.yml b/.github/workflows/build-relocatable-packages.yml new file mode 100644 index 000000000..1e9e7041e --- /dev/null +++ b/.github/workflows/build-relocatable-packages.yml @@ -0,0 +1,998 @@ +name: Build Relocatable Packages + +on: + push: + branches: + - master + - main + - 'release/**' + pull_request: + branches: + - master + - main + schedule: + # Run daily at 5:00 AM PST (13:00 UTC; PST is UTC-8) + - cron: '0 10 * * *' + workflow_dispatch: + inputs: + rocm_version: + description: 'Optional pin — empty uses latest nightly (unless vars.ROCM_VERSION set). Release X.Y.Z vs nightly x.y.za… selects tarball host by format.' + required: false + default: '' + gpu_family: + description: 'GPU family target' + required: false + type: choice + options: + - gfx94X-dcgpu + - gfx950-dcgpu + - gfx110X-all + - gfx1151 + - gfx120X-all + default: 'gfx110X-all' + build_transferbench_cli: + description: 'Build and bundle TransferBench CLI in DEB/RPM/TGZ' + required: false + type: boolean + default: true + +permissions: + contents: read + id-token: write + +env: + # Leave unset unless you pin a version via repo variable or workflow_dispatch input. + # With vars.ROCM_SDK_RELEASE_URL set, the script picks latest X.Y.Z from that listing for GPU_FAMILY. + ROCM_VERSION: ${{ vars.ROCM_VERSION }} + GPU_FAMILY: ${{ vars.GPU_FAMILY || 'gfx110X-all' }} + BUILD_TYPE: Release + # TransferBench CLI: OFF in build_packages_local.sh by default; ON in CI unless overridden. + BUILD_TRANSFERBENCH_CLI: ${{ github.event_name == 'workflow_dispatch' && (github.event.inputs.build_transferbench_cli && 'ON' || 'OFF') || vars.BUILD_TRANSFERBENCH_CLI || 'ON' }} + # Optional Step 6 in build_packages_local.sh: set repo Actions variable + # RVS_AUTO_DETECT_LOCAL_UPLOAD to "1" to probe localhost:8080 when UPLOAD_TARGET is unset. + # Leave the variable undefined for default (no probe; S3 uses workflow steps only). + RVS_AUTO_DETECT_LOCAL_UPLOAD: ${{ vars.RVS_AUTO_DETECT_LOCAL_UPLOAD }} + # Optional local package server upload (Step 6 of build_packages_local.sh). + # Disabled by default. To enable, set repo Actions variable + # LOCAL_PACKAGE_SERVER_URL to a reachable HTTP server that accepts PUT. + # S3 upload steps also run when repository variable RVS_S3_UPLOAD_ENABLED is the string 'true' + # (in addition to the upstream ROCm/ROCmValidationSuite repo). See README_BUILD_PACKAGES.md. + +jobs: + prepare-nightly-branches: + name: Prepare nightly branch matrix + runs-on: ${{ vars.RUNNER_LABEL_UTILITY || 'ubuntu-latest' }} + outputs: + branches_json: ${{ steps.branches.outputs.branches_json }} + steps: + - name: Resolve branch matrix + id: branches + env: + ACTIVE_BRANCHES: ${{ vars.ACTIVE_BRANCHES }} + DEFAULT_BRANCH: ${{ github.event.repository.default_branch }} + EVENT_NAME: ${{ github.event_name }} + REF_NAME: ${{ github.ref_name }} + PR_HEAD_REF: ${{ github.event.pull_request.head.ref }} + GITHUB_TOKEN: ${{ github.token }} + GITHUB_REPOSITORY: ${{ github.repository }} + run: | + python3 <<'PY' + import json + import os + import re + import urllib.request + + def list_remote_branches(): + token = os.environ["GITHUB_TOKEN"] + repo = os.environ["GITHUB_REPOSITORY"] + names = [] + page = 1 + while True: + url = ( + f"https://api.github.com/repos/{repo}/branches" + f"?per_page=100&page={page}" + ) + req = urllib.request.Request( + url, + headers={ + "Authorization": f"Bearer {token}", + "Accept": "application/vnd.github+json", + "User-Agent": "rvs-build-relocatable-packages", + }, + ) + with urllib.request.urlopen(req) as resp: + batch = json.loads(resp.read().decode()) + if not batch: + break + for item in batch: + names.append(item["name"]) + if len(batch) < 100: + break + page += 1 + return names + + def branch_prefix(name: str) -> str: + return name.split("/", 1)[0] if "/" in name else name + + def glob_match(name: str, pattern: str) -> bool: + regex_parts = [] + i = 0 + while i < len(pattern): + if pattern[i : i + 2] == "**": + regex_parts.append(".*") + i += 2 + elif pattern[i] == "*": + regex_parts.append("[^/]*") + i += 1 + else: + regex_parts.append(re.escape(pattern[i])) + i += 1 + return re.fullmatch("".join(regex_parts), name) is not None + + def skip_upload_schedule(branch: str) -> bool: + return branch == "release" or branch.startswith("release/") + + event = os.environ["EVENT_NAME"] + default = os.environ["DEFAULT_BRANCH"] + + if event != "schedule": + if event == "pull_request": + branch = os.environ.get("PR_HEAD_REF") or os.environ["REF_NAME"] + else: + branch = os.environ["REF_NAME"] + branches = [{ + "branch": branch, + "is_default": branch == default, + "skip_upload": False, + "branch_prefix": branch_prefix(branch), + }] + else: + patterns = [ + p.strip() + for p in os.environ.get("ACTIVE_BRANCHES", "").split(",") + if p.strip() + ] + remote_names = list_remote_branches() + + matched = set() + for pattern in patterns: + for name in remote_names: + if glob_match(name, pattern) and name != default: + matched.add(name) + + branches = [{ + "branch": default, + "is_default": True, + "skip_upload": False, + "branch_prefix": branch_prefix(default), + }] + for name in sorted(matched): + branches.append({ + "branch": name, + "is_default": False, + "skip_upload": skip_upload_schedule(name), + "branch_prefix": branch_prefix(name), + }) + + payload = json.dumps(branches) + print(payload) + with open(os.environ["GITHUB_OUTPUT"], "a", encoding="utf-8") as out: + out.write(f"branches_json={payload}\n") + PY + + # Runs on the host (no job container) so we can reclaim runner disk before Docker + # image pulls and ROCm SDK extracts. Failures here are often "No space left on device". + runner-disk-prep: + name: Free runner disk space + runs-on: ${{ vars.RUNNER_LABEL || 'ubuntu-22.04' }} + steps: + # No checkout. Container jobs leave root-owned files in the persistent + # workspace, and this host step cannot lock or delete them. + - name: Free disk space + run: | + set -u + as_root() { + if [ "$(id -u)" -eq 0 ]; then + "$@" + elif command -v sudo >/dev/null 2>&1; then + sudo -n "$@" + else + echo "::warning::not root and sudo is unavailable; skipped: $*" + return 0 + fi + } + echo "=== Disk before cleanup ===" + df -h / || df -h || true + if command -v apt-get >/dev/null 2>&1; then + echo "apt-get clean" + as_root apt-get clean || echo "::warning::apt-get clean did not succeed" + else + echo "skip: apt-get not available" + fi + if command -v docker >/dev/null 2>&1; then + echo "pruning unused Docker images" + if ! docker image prune --all --force; then + as_root docker image prune --all --force \ + || echo "::warning::docker image prune did not succeed" + fi + else + echo "skip: docker not installed" + fi + echo "=== Disk after cleanup ===" + df -h / || df -h || true + exit 0 + + - name: Prune Docker and report disk usage + run: | + echo "=== Disk before prune ===" + df -h / || df -h + if command -v docker >/dev/null 2>&1; then + docker system prune -af --volumes || true + fi + echo "=== Disk after prune ===" + df -h / || df -h + + build-ubuntu: + name: Build Ubuntu Packages (${{ matrix.branch }}) + needs: [prepare-nightly-branches, runner-disk-prep] + runs-on: ${{ vars.RUNNER_LABEL || 'ubuntu-22.04' }} + container: + image: ubuntu:22.04 + # Allow the build container to reach the runner host via host.docker.internal. + options: --add-host=host.docker.internal:host-gateway + strategy: + fail-fast: false + matrix: + include: ${{ fromJson(needs.prepare-nightly-branches.outputs.branches_json) }} + # Apply to every step in the job so apt never drops into an interactive + # frontend (tzdata, libssl, etc.) — each Actions step runs in a fresh + # shell, so an `export` inside one step does not propagate to the next. + env: + DEBIAN_FRONTEND: noninteractive + TZ: Etc/UTC + outputs: + s3_bucket_ubuntu: ${{ steps.upload-s3-ubuntu.outputs.bucket }} + s3_paths_ubuntu: ${{ steps.upload-s3-ubuntu.outputs.paths }} + rocm_version: ${{ steps.capture-rocm-version.outputs.rocm_version }} + + steps: + # Bootstrap the bare ubuntu:* image so actions/checkout and downstream steps + # (dpkg-deb verify, apt-ftparchive, awscli) work without needing sudo on bare metal. + - name: Bootstrap container (git, curl, packaging tools) + run: | + export DEBIAN_FRONTEND=noninteractive + apt-get update + apt-get install -y --no-install-recommends \ + ca-certificates \ + curl \ + git \ + dpkg-dev \ + apt-utils \ + reprepro \ + python3 \ + python3-pip + rm -rf /var/lib/apt/lists/* + + - name: Checkout Repository + uses: actions/checkout@v4 + with: + ref: ${{ matrix.branch }} + submodules: recursive + fetch-depth: 0 + + - name: Set build branch context + run: | + echo "RVS_BUILD_REF_NAME=${{ matrix.branch }}" >> "$GITHUB_ENV" + echo "RVS_IS_DEFAULT_BRANCH=${{ matrix.is_default }}" >> "$GITHUB_ENV" + echo "RVS_SKIP_UPLOAD=${{ matrix.skip_upload }}" >> "$GITHUB_ENV" + echo "RVS_BRANCH_PREFIX=${{ matrix.branch_prefix }}" >> "$GITHUB_ENV" + + # Schedule → nightly SDK. PR/push main|release/* → release SDK. Manual pin → format (auto). + - name: Configure ROCm SDK channel and version + run: bash .github/scripts/configure-rocm-sdk-channel.sh + env: + INPUT_ROCM_VERSION: ${{ github.event.inputs.rocm_version }} + VAR_ROCM_VERSION: ${{ vars.ROCM_VERSION }} + INPUT_GPU_FAMILY: ${{ github.event.inputs.gpu_family }} + ROCM_SDK_RELEASE_URL: ${{ vars.ROCM_SDK_RELEASE_URL }} + ROCM_SDK_RELEASE_BASE_URL: ${{ vars.ROCM_SDK_RELEASE_BASE_URL }} + ROCM_SDK_NIGHTLY_BASE_URL: ${{ vars.ROCM_SDK_NIGHTLY_BASE_URL }} + ROCM_SDK_NIGHTLY_INDEX_URL: ${{ vars.ROCM_SDK_NIGHTLY_INDEX_URL }} + + # Forward LOCAL_PACKAGE_SERVER_URL to the build script as UPLOAD_TARGET. + - name: Configure local package upload + if: vars.LOCAL_PACKAGE_SERVER_URL != '' + run: | + echo "UPLOAD_TARGET=${{ vars.LOCAL_PACKAGE_SERVER_URL }}" >> "$GITHUB_ENV" + + - name: Build and Package RVS + run: | + chmod +x build_packages_local.sh + # Container runs as root; ROCM_SDK_* come from "Configure ROCm SDK channel" via GITHUB_ENV. + ./build_packages_local.sh + + - name: Capture resolved ROCm version + id: capture-rocm-version + run: | + echo "rocm_version=${ROCM_VERSION}" >> "$GITHUB_OUTPUT" + + # Print the resulting download URL in the step log and run Summary. + - name: Print local package download URL + if: vars.LOCAL_PACKAGE_SERVER_URL != '' + run: | + REPO="${GITHUB_REPOSITORY##*/}" + BRANCH="${RVS_BUILD_REF_NAME:-$GITHUB_REF_NAME}" + SAFE_BRANCH=$(echo "$BRANCH" | sed 's|[^a-zA-Z0-9._/-]|-|g') + DATE=$(date +%Y-%m-%d) + URL="${{ vars.LOCAL_PACKAGE_SERVER_URL }}/${REPO}/${SAFE_BRANCH}/${DATE}/" + echo "================================================================" + echo " Package downloads for this build (Ubuntu)" + echo " $URL" + echo "================================================================" + { + echo "## Package Downloads (Ubuntu)" + echo "" + echo "[$URL]($URL)" + } >> "$GITHUB_STEP_SUMMARY" + + - name: Remove ROCm SDK tree after build + if: always() + run: | + rm -rf "$HOME/rocm-sdk" 2>/dev/null || true + df -h / || df -h + + + - name: Verify DEB Package + run: | + cd ./build + DEB_FILE=$(ls amdrocm*-rvs*.deb | head -1) + if [ -n "$DEB_FILE" ]; then + echo "=== DEB Package Info ===" + dpkg-deb -I "$DEB_FILE" + echo "" + echo "=== DEB Package Contents (first 30 files) ===" + dpkg-deb -c "$DEB_FILE" | head -30 + echo "" + echo "Package filename: $DEB_FILE" + fi + + # S3 upload: upstream repo, or any repo with vars.RVS_S3_UPLOAD_ENABLED == 'true' (skip fork PRs: OIDC empty → JSON parse fails) + - name: Install AWS CLI + if: (github.repository == 'ROCm/ROCmValidationSuite' || vars.RVS_S3_UPLOAD_ENABLED == 'true') && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository) && matrix.skip_upload != true + run: apt-get update && apt-get install -y awscli + + - name: Configure AWS credentials + if: (github.repository == 'ROCm/ROCmValidationSuite' || vars.RVS_S3_UPLOAD_ENABLED == 'true') && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository) && matrix.skip_upload != true + run: | + TOKEN=$(curl -s -H "Authorization: bearer $ACTIONS_ID_TOKEN_REQUEST_TOKEN" \ + "${ACTIONS_ID_TOKEN_REQUEST_URL}&audience=sts.amazonaws.com" | python3 -c "import sys,json; print(json.load(sys.stdin)['value'])") + CREDS=$(aws sts assume-role-with-web-identity \ + --role-arn "${{ secrets.AWS_ROLE_ARN }}" \ + --role-session-name "github-actions-${{ github.run_id }}" \ + --web-identity-token "$TOKEN" \ + --query 'Credentials' --output json) + echo "AWS_ACCESS_KEY_ID=$(echo "$CREDS" | python3 -c "import sys,json; print(json.load(sys.stdin)['AccessKeyId'])")" >> $GITHUB_ENV + echo "AWS_SECRET_ACCESS_KEY=$(echo "$CREDS" | python3 -c "import sys,json; print(json.load(sys.stdin)['SecretAccessKey'])")" >> $GITHUB_ENV + echo "AWS_SESSION_TOKEN=$(echo "$CREDS" | python3 -c "import sys,json; print(json.load(sys.stdin)['SessionToken'])")" >> $GITHUB_ENV + echo "AWS_DEFAULT_REGION=us-east-1" >> $GITHUB_ENV + echo "::add-mask::$(echo "$CREDS" | python3 -c "import sys,json; print(json.load(sys.stdin)['SecretAccessKey'])")" + echo "::add-mask::$(echo "$CREDS" | python3 -c "import sys,json; print(json.load(sys.stdin)['SessionToken'])")" + env: + AWS_DEFAULT_REGION: us-east-1 + + # S3 routing: POSIX script (ubuntu:22.04 container uses sh — no bash [[). + - name: Upload Ubuntu packages to S3 + id: upload-s3-ubuntu + if: (github.repository == 'ROCm/ROCmValidationSuite' || vars.RVS_S3_UPLOAD_ENABLED == 'true') && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository) && matrix.skip_upload != true + run: sh .github/scripts/rvs-s3-upload-route.sh upload-deb + env: + AWS_S3_BUCKET: ${{ vars.AWS_S3_BUCKET }} + RVS_RUNNER_SUFFIX: ubuntu-22.04 + + # Generate APT repo metadata (Packages, Packages.gz, Release) so the S3 path + # can be consumed directly as a DEB repository by apt. + # Only runs when S3 upload is allowed (upstream or RVS_S3_UPLOAD_ENABLED) and the event is schedule, push, or manual. + - name: Generate and upload DEB repo metadata to S3 + if: (github.repository == 'ROCm/ROCmValidationSuite' || vars.RVS_S3_UPLOAD_ENABLED == 'true') && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository) && (github.event_name == 'schedule' || github.event_name == 'push' || github.event_name == 'workflow_dispatch') && matrix.skip_upload != true && (github.event_name != 'schedule' || matrix.is_default == true) + run: | + BUCKET="${{ vars.AWS_S3_BUCKET }}" + [ -z "$BUCKET" ] && exit 0 + + # dpkg-dev and apt-utils were installed in the Bootstrap container step. + set -- $(sh .github/scripts/rvs-s3-upload-route.sh deb-repo-prefix) + DEB_PREFIX="$1" + APT_SUITE="$2" + + STAGING=$(mktemp -d) + + echo "Downloading existing DEBs from s3://${BUCKET}/${DEB_PREFIX}/ ..." + aws s3 sync "s3://${BUCKET}/${DEB_PREFIX}/" "$STAGING/" \ + --exclude "Packages*" --exclude "Release*" --exclude "InRelease" \ + --no-progress + + cp ./build/amdrocm*-rvs*.deb "$STAGING/" 2>/dev/null || true + + cd "$STAGING" + dpkg-scanpackages --multiversion . /dev/null > Packages + # dpkg-scanpackages emits "Filename: ./foo.deb". APT resolves that under the repo URL and may + # request .../deb/./foo.deb — S3/CloudFront keys are literal, so that path 404s. Strip "./". + sed -i 's#^Filename: \./#Filename: #g' Packages + gzip -k -f Packages + apt-ftparchive \ + -o APT::FTPArchive::Release::Origin="ROCm Validation Suite" \ + -o APT::FTPArchive::Release::Label="${APT_SUITE}" \ + -o APT::FTPArchive::Release::Suite="${APT_SUITE}" \ + -o APT::FTPArchive::Release::Codename="${APT_SUITE}" \ + release . > Release + + aws s3 sync "$STAGING/" "s3://${BUCKET}/${DEB_PREFIX}/" --no-progress + + echo "=== DEB repo metadata uploaded to s3://${BUCKET}/${DEB_PREFIX}/ ===" + aws s3 ls "s3://${BUCKET}/${DEB_PREFIX}/" --human-readable || true + env: + AWS_S3_BUCKET: ${{ vars.AWS_S3_BUCKET }} + RVS_RUNNER_SUFFIX: ubuntu-22.04 + + # Scheduled default-branch only: unsigned DEB archive (dists/ + pool/) for signing CI. + - name: Generate and upload unsigned DEB archive to S3 + if: (github.repository == 'ROCm/ROCmValidationSuite' || vars.RVS_S3_UPLOAD_ENABLED == 'true') && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository) && matrix.skip_upload != true && github.event_name == 'schedule' && matrix.is_default == true + run: | + BUCKET="${{ vars.AWS_S3_BUCKET }}" + [ -z "$BUCKET" ] && exit 0 + export DEBIAN_FRONTEND=noninteractive + chmod +x .github/scripts/rvs-deb-unsigned-repo.sh + sh .github/scripts/rvs-deb-unsigned-repo.sh ./build + env: + AWS_S3_BUCKET: ${{ vars.AWS_S3_BUCKET }} + GITHUB_RUN_ID: ${{ github.run_id }} + RVS_RUNNER_SUFFIX: ubuntu-22.04 + + build-centos: + name: Build CentOS/RHEL Packages (${{ matrix.branch }}) + needs: [prepare-nightly-branches, runner-disk-prep] + runs-on: ${{ vars.RUNNER_LABEL_CONTAINER || 'ubuntu-latest' }} + container: + image: quay.io/pypa/manylinux_2_28_x86_64 + # Allow the build container to reach the runner host via host.docker.internal. + options: --add-host=host.docker.internal:host-gateway + strategy: + fail-fast: false + matrix: + include: ${{ fromJson(needs.prepare-nightly-branches.outputs.branches_json) }} + outputs: + s3_bucket_centos: ${{ steps.upload-s3-centos.outputs.bucket }} + s3_paths_centos: ${{ steps.upload-s3-centos.outputs.paths }} + + steps: + - name: Install Git (for checkout) + run: | + yum install -y git + + - name: Checkout Repository + uses: actions/checkout@v4 + with: + ref: ${{ matrix.branch }} + submodules: recursive + fetch-depth: 0 + + - name: Set build branch context + run: | + echo "RVS_BUILD_REF_NAME=${{ matrix.branch }}" >> "$GITHUB_ENV" + echo "RVS_IS_DEFAULT_BRANCH=${{ matrix.is_default }}" >> "$GITHUB_ENV" + echo "RVS_SKIP_UPLOAD=${{ matrix.skip_upload }}" >> "$GITHUB_ENV" + echo "RVS_BRANCH_PREFIX=${{ matrix.branch_prefix }}" >> "$GITHUB_ENV" + + - name: Configure ROCm SDK channel and version + run: bash .github/scripts/configure-rocm-sdk-channel.sh + env: + INPUT_ROCM_VERSION: ${{ github.event.inputs.rocm_version }} + VAR_ROCM_VERSION: ${{ vars.ROCM_VERSION }} + INPUT_GPU_FAMILY: ${{ github.event.inputs.gpu_family }} + ROCM_SDK_RELEASE_URL: ${{ vars.ROCM_SDK_RELEASE_URL }} + ROCM_SDK_RELEASE_BASE_URL: ${{ vars.ROCM_SDK_RELEASE_BASE_URL }} + ROCM_SDK_NIGHTLY_BASE_URL: ${{ vars.ROCM_SDK_NIGHTLY_BASE_URL }} + ROCM_SDK_NIGHTLY_INDEX_URL: ${{ vars.ROCM_SDK_NIGHTLY_INDEX_URL }} + + - name: Configure local package upload + if: vars.LOCAL_PACKAGE_SERVER_URL != '' + run: | + echo "UPLOAD_TARGET=${{ vars.LOCAL_PACKAGE_SERVER_URL }}" >> "$GITHUB_ENV" + + - name: Build and Package RVS + run: | + chmod +x build_packages_local.sh + ./build_packages_local.sh + + - name: Print local package download URL + if: vars.LOCAL_PACKAGE_SERVER_URL != '' + run: | + REPO="${GITHUB_REPOSITORY##*/}" + BRANCH="${RVS_BUILD_REF_NAME:-$GITHUB_REF_NAME}" + SAFE_BRANCH=$(echo "$BRANCH" | sed 's|[^a-zA-Z0-9._/-]|-|g') + DATE=$(date +%Y-%m-%d) + URL="${{ vars.LOCAL_PACKAGE_SERVER_URL }}/${REPO}/${SAFE_BRANCH}/${DATE}/" + echo "================================================================" + echo " Package downloads for this build (CentOS/RHEL)" + echo " $URL" + echo "================================================================" + { + echo "## Package Downloads (CentOS/RHEL)" + echo "" + echo "[$URL]($URL)" + } >> "$GITHUB_STEP_SUMMARY" + + - name: Verify RPM Package + run: | + cd ./build + RPM_FILE=$(ls amdrocm*-rvs*.rpm | head -1) + if [ -n "$RPM_FILE" ]; then + echo "=== RPM Package Info ===" + rpm -qip "$RPM_FILE" + echo "" + echo "=== RPM Package Contents (first 30 files) ===" + rpm -qlp "$RPM_FILE" | head -30 + echo "" + echo "=== RPM Dependencies ===" + rpm -qRp "$RPM_FILE" + echo "" + echo "Package filename: $RPM_FILE" + fi + + # S3 upload: upstream repo, or RVS_S3_UPLOAD_ENABLED (skip fork PRs — OIDC not usable) + - name: Install AWS CLI + if: (github.repository == 'ROCm/ROCmValidationSuite' || vars.RVS_S3_UPLOAD_ENABLED == 'true') && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository) && matrix.skip_upload != true + run: | + python3 -m pip install --break-system-packages awscli || (yum install -y python3-pip && pip3 install awscli) + + - name: Configure AWS credentials + if: (github.repository == 'ROCm/ROCmValidationSuite' || vars.RVS_S3_UPLOAD_ENABLED == 'true') && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository) && matrix.skip_upload != true + run: | + TOKEN=$(curl -s -H "Authorization: bearer $ACTIONS_ID_TOKEN_REQUEST_TOKEN" \ + "${ACTIONS_ID_TOKEN_REQUEST_URL}&audience=sts.amazonaws.com" | python3 -c "import sys,json; print(json.load(sys.stdin)['value'])") + CREDS=$(aws sts assume-role-with-web-identity \ + --role-arn "${{ secrets.AWS_ROLE_ARN }}" \ + --role-session-name "github-actions-${{ github.run_id }}" \ + --web-identity-token "$TOKEN" \ + --query 'Credentials' --output json) + echo "AWS_ACCESS_KEY_ID=$(echo "$CREDS" | python3 -c "import sys,json; print(json.load(sys.stdin)['AccessKeyId'])")" >> $GITHUB_ENV + echo "AWS_SECRET_ACCESS_KEY=$(echo "$CREDS" | python3 -c "import sys,json; print(json.load(sys.stdin)['SecretAccessKey'])")" >> $GITHUB_ENV + echo "AWS_SESSION_TOKEN=$(echo "$CREDS" | python3 -c "import sys,json; print(json.load(sys.stdin)['SessionToken'])")" >> $GITHUB_ENV + echo "AWS_DEFAULT_REGION=us-east-1" >> $GITHUB_ENV + echo "::add-mask::$(echo "$CREDS" | python3 -c "import sys,json; print(json.load(sys.stdin)['SecretAccessKey'])")" + echo "::add-mask::$(echo "$CREDS" | python3 -c "import sys,json; print(json.load(sys.stdin)['SessionToken'])")" + env: + AWS_DEFAULT_REGION: us-east-1 + + - name: Upload CentOS/RHEL packages to S3 + id: upload-s3-centos + if: (github.repository == 'ROCm/ROCmValidationSuite' || vars.RVS_S3_UPLOAD_ENABLED == 'true') && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository) && matrix.skip_upload != true + run: sh .github/scripts/rvs-s3-upload-route.sh upload-rpm-tar + env: + AWS_S3_BUCKET: ${{ vars.AWS_S3_BUCKET }} + RVS_RUNNER_SUFFIX: manylinux_2_28 + + - name: Generate SHA-256 sidecars for TGZ packages + if: (github.repository == 'ROCm/ROCmValidationSuite' || vars.RVS_S3_UPLOAD_ENABLED == 'true') && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository) && matrix.skip_upload != true && github.event_name == 'schedule' && matrix.is_default == true + run: | + cd ./build + for tar in amdrocm*-rvs*.tar.gz; do + [ -f "$tar" ] || continue + sha256sum "$tar" > "${tar}.sha256" + echo "Wrote ${tar}.sha256" + done + + - name: Upload unsigned CentOS/RHEL packages to S3 + if: (github.repository == 'ROCm/ROCmValidationSuite' || vars.RVS_S3_UPLOAD_ENABLED == 'true') && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository) && matrix.skip_upload != true && github.event_name == 'schedule' && matrix.is_default == true + run: sh .github/scripts/rvs-s3-upload-route.sh upload-rpm-tar-unsigned + env: + AWS_S3_BUCKET: ${{ vars.AWS_S3_BUCKET }} + RVS_RUNNER_SUFFIX: manylinux_2_28 + + # Generate YUM/DNF repo metadata (repodata/) so the S3 path can be consumed + # directly as an RPM repository by yum/dnf. + # Only runs when S3 upload is allowed (upstream or RVS_S3_UPLOAD_ENABLED) and the event is schedule, push, or manual. + - name: Generate and upload RPM repodata to S3 + if: (github.repository == 'ROCm/ROCmValidationSuite' || vars.RVS_S3_UPLOAD_ENABLED == 'true') && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository) && (github.event_name == 'schedule' || github.event_name == 'push' || github.event_name == 'workflow_dispatch') && matrix.skip_upload != true && (github.event_name != 'schedule' || matrix.is_default == true) + run: | + BUCKET="${{ vars.AWS_S3_BUCKET }}" + [ -z "$BUCKET" ] && exit 0 + + yum install -y createrepo_c || yum install -y createrepo + + RPM_PREFIX=$(sh .github/scripts/rvs-s3-upload-route.sh rpm-repo-prefix) + + STAGING=$(mktemp -d) + + echo "Downloading existing RPMs from s3://${BUCKET}/${RPM_PREFIX}/ ..." + aws s3 sync "s3://${BUCKET}/${RPM_PREFIX}/" "$STAGING/" \ + --exclude "repodata/*" --no-progress + + cp ./build/amdrocm*-rvs*.rpm "$STAGING/" 2>/dev/null || true + + createrepo_c --simple-md-filenames --no-database --compress-type gz "$STAGING" \ + || createrepo "$STAGING" + + aws s3 sync "$STAGING/" "s3://${BUCKET}/${RPM_PREFIX}/" --no-progress + + echo "=== RPM repodata uploaded to s3://${BUCKET}/${RPM_PREFIX}/repodata/ ===" + aws s3 ls "s3://${BUCKET}/${RPM_PREFIX}/repodata/" --human-readable || true + env: + AWS_S3_BUCKET: ${{ vars.AWS_S3_BUCKET }} + RVS_RUNNER_SUFFIX: manylinux_2_28 + + - name: Generate and upload unsigned RPM repodata to S3 + if: (github.repository == 'ROCm/ROCmValidationSuite' || vars.RVS_S3_UPLOAD_ENABLED == 'true') && (github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository) && matrix.skip_upload != true && github.event_name == 'schedule' && matrix.is_default == true + run: | + BUCKET="${{ vars.AWS_S3_BUCKET }}" + [ -z "$BUCKET" ] && exit 0 + + rpm_n=0 + for f in ./build/amdrocm*-rvs*.rpm; do + [ -f "$f" ] || continue + rpm_n=$((rpm_n + 1)) + done + tar_n=0 + for f in ./build/amdrocm*-rvs*.tar.gz; do + [ -f "$f" ] || continue + tar_n=$((tar_n + 1)) + if [ ! -f "${f}.sha256" ]; then + echo "::error::Missing SHA-256 sidecar for $(basename "$f")." >&2 + exit 1 + fi + done + if [ "$rpm_n" -lt 1 ] || [ "$tar_n" -lt 1 ]; then + echo "::error::Unsigned repodata requires amdrocm*-rvs*.rpm and amdrocm*-rvs*.tar.gz in ./build." >&2 + exit 1 + fi + + yum install -y createrepo_c || yum install -y createrepo + + RPM_PREFIX=$(sh .github/scripts/rvs-s3-upload-route.sh unsigned-rpm-prefix) + TAR_PREFIX=$(sh .github/scripts/rvs-s3-upload-route.sh unsigned-tar-prefix) + export RPM_PREFIX TAR_PREFIX + + STAGING=$(mktemp -d) + + echo "Downloading existing unsigned RPMs from s3://${BUCKET}/${RPM_PREFIX}/ (accumulate) ..." + aws s3 sync "s3://${BUCKET}/${RPM_PREFIX}/" "$STAGING/" \ + --exclude "repodata/*" --no-progress + + cp ./build/amdrocm*-rvs*.rpm "$STAGING/" + + createrepo_c --simple-md-filenames --no-database --compress-type gz "$STAGING" \ + || createrepo "$STAGING" + + aws s3 sync "$STAGING/" "s3://${BUCKET}/${RPM_PREFIX}/" --no-progress + + echo "=== Unsigned RPM repodata uploaded to s3://${BUCKET}/${RPM_PREFIX}/repodata/ ===" + aws s3 ls "s3://${BUCKET}/${RPM_PREFIX}/repodata/" --human-readable || true + + python3 <<'PY' + import hashlib + import json + import os + from pathlib import Path + + build = Path("build") + rpm_prefix = os.environ["RPM_PREFIX"] + tar_prefix = os.environ["TAR_PREFIX"] + run_id = os.environ["GITHUB_RUN_ID"] + + def sha256_file(path: Path) -> str: + h = hashlib.sha256() + with path.open("rb") as f: + for chunk in iter(lambda: f.read(1024 * 1024), b""): + h.update(chunk) + return h.hexdigest() + + import sys + + rpms = sorted(build.glob("amdrocm*-rvs*.rpm")) + tars = sorted(build.glob("amdrocm*-rvs*.tar.gz")) + if not rpms or not tars: + print("::error::Missing RPM or TGZ for unsigned rpm-tar.json", file=sys.stderr) + sys.exit(1) + + rpm = rpms[0] + tar = tars[0] + sidecar = Path(str(tar) + ".sha256") + if not sidecar.is_file(): + print(f"::error::Missing SHA-256 sidecar for {tar.name}", file=sys.stderr) + sys.exit(1) + + payload = { + "github_run_id": run_id, + "rpm": { + "filename": rpm.name, + "s3_key": f"{rpm_prefix}/{rpm.name}", + "sha256": sha256_file(rpm), + }, + "tar": { + "filename": tar.name, + "s3_key": f"{tar_prefix}/{tar.name}", + "sha256": sha256_file(tar), + "sha256_sidecar_key": f"{tar_prefix}/{sidecar.name}", + }, + } + + meta = Path("build/unsigned-rpm-tar-meta.json") + meta.write_text(json.dumps(payload, indent=2) + "\n", encoding="utf-8") + print(meta.read_text(encoding="utf-8")) + PY + + aws s3 cp build/unsigned-rpm-tar-meta.json \ + "s3://${BUCKET}/nightly/unsigned/runs/${GITHUB_RUN_ID}/rpm-tar.json" --no-progress + env: + AWS_S3_BUCKET: ${{ vars.AWS_S3_BUCKET }} + GITHUB_RUN_ID: ${{ github.run_id }} + RVS_RUNNER_SUFFIX: manylinux_2_28 + + publish-unsigned-latest: + name: Publish unsigned latest.json (signing CI) + needs: [build-ubuntu, build-centos] + runs-on: ${{ vars.RUNNER_LABEL_UTILITY || 'ubuntu-latest' }} + if: >- + github.event_name == 'schedule' && + needs.build-ubuntu.result == 'success' && + needs.build-centos.result == 'success' && + (github.repository == 'ROCm/ROCmValidationSuite' || vars.RVS_S3_UPLOAD_ENABLED == 'true') && + vars.AWS_S3_BUCKET != '' + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + + - name: Install AWS CLI + if: github.repository == 'ROCm/ROCmValidationSuite' || vars.RVS_S3_UPLOAD_ENABLED == 'true' + run: | + # ubuntu-latest (24.04+) no longer ships awscli in apt; use pip like the CentOS job. + if command -v aws >/dev/null 2>&1; then + aws --version + exit 0 + fi + python3 -m pip install --user --break-system-packages awscli \ + || python3 -m pip install --user awscli + echo "$HOME/.local/bin" >> "$GITHUB_PATH" + # GITHUB_PATH only applies to later steps; export for this step's verify. + export PATH="$HOME/.local/bin:$PATH" + aws --version + + - name: Configure AWS credentials + if: github.repository == 'ROCm/ROCmValidationSuite' || vars.RVS_S3_UPLOAD_ENABLED == 'true' + run: | + TOKEN=$(curl -s -H "Authorization: bearer $ACTIONS_ID_TOKEN_REQUEST_TOKEN" \ + "${ACTIONS_ID_TOKEN_REQUEST_URL}&audience=sts.amazonaws.com" | python3 -c "import sys,json; print(json.load(sys.stdin)['value'])") + CREDS=$(aws sts assume-role-with-web-identity \ + --role-arn "${{ secrets.AWS_ROLE_ARN }}" \ + --role-session-name "github-actions-${{ github.run_id }}" \ + --web-identity-token "$TOKEN" \ + --query 'Credentials' --output json) + echo "AWS_ACCESS_KEY_ID=$(echo "$CREDS" | python3 -c "import sys,json; print(json.load(sys.stdin)['AccessKeyId'])")" >> $GITHUB_ENV + echo "AWS_SECRET_ACCESS_KEY=$(echo "$CREDS" | python3 -c "import sys,json; print(json.load(sys.stdin)['SecretAccessKey'])")" >> $GITHUB_ENV + echo "AWS_SESSION_TOKEN=$(echo "$CREDS" | python3 -c "import sys,json; print(json.load(sys.stdin)['SessionToken'])")" >> $GITHUB_ENV + echo "AWS_DEFAULT_REGION=us-east-1" >> $GITHUB_ENV + echo "::add-mask::$(echo "$CREDS" | python3 -c "import sys,json; print(json.load(sys.stdin)['SecretAccessKey'])")" + echo "::add-mask::$(echo "$CREDS" | python3 -c "import sys,json; print(json.load(sys.stdin)['SessionToken'])")" + env: + AWS_DEFAULT_REGION: us-east-1 + + - name: Publish nightly/unsigned/latest.json + if: github.repository == 'ROCm/ROCmValidationSuite' || vars.RVS_S3_UPLOAD_ENABLED == 'true' + run: | + BUCKET="${{ vars.AWS_S3_BUCKET }}" + [ -z "$BUCKET" ] && exit 0 + chmod +x .github/scripts/rvs-unsigned-publish-latest.sh + sh .github/scripts/rvs-unsigned-publish-latest.sh + env: + AWS_S3_BUCKET: ${{ vars.AWS_S3_BUCKET }} + GITHUB_RUN_ID: ${{ github.run_id }} + GITHUB_SHA: ${{ github.sha }} + ROCM_VERSION: ${{ needs.build-ubuntu.outputs.rocm_version }} + + dispatch-post-build-tests: + name: Dispatch post-build test workflows + needs: [build-ubuntu, build-centos] + runs-on: ${{ vars.RUNNER_LABEL_UTILITY || 'ubuntu-latest' }} + if: >- + always() && !cancelled() && + needs.build-ubuntu.result == 'success' && + needs.build-centos.result == 'success' + permissions: + actions: write + contents: read + steps: + - name: Dispatch RVS Nightly Tests + if: >- + github.event_name == 'schedule' || + (github.event_name == 'push' && (github.ref_name == 'main' || github.ref_name == 'master')) + uses: actions/github-script@v7 + with: + script: | + const ref = context.payload.pull_request?.base?.ref + ?? context.ref.replace('refs/heads/', ''); + const inputs = { trigger_event: context.eventName }; + try { + await github.rest.actions.createWorkflowDispatch({ + owner: context.repo.owner, + repo: context.repo.repo, + workflow_id: 'rvs-nightly-tests.yml', + ref, + inputs, + }); + } catch (error) { + if (error.status === 422 && String(error.message).includes('Unexpected inputs')) { + core.warning('rvs-nightly-tests.yml on ref does not define trigger_event — dispatching without it.'); + await github.rest.actions.createWorkflowDispatch({ + owner: context.repo.owner, + repo: context.repo.repo, + workflow_id: 'rvs-nightly-tests.yml', + ref, + inputs: {}, + }); + } else { + throw error; + } + } + core.info(`Dispatched RVS Nightly Tests (trigger_event=${context.eventName}) on ref ${ref}`); + + - name: Dispatch RVS Nightly Docker Tests (Ubuntu 24.04) + if: >- + github.event_name == 'push' && + (github.ref_name == 'main' || github.ref_name == 'master') + uses: actions/github-script@v7 + with: + script: | + const ref = context.payload.pull_request?.base?.ref + ?? context.ref.replace('refs/heads/', ''); + const inputs = { trigger_event: context.eventName }; + try { + await github.rest.actions.createWorkflowDispatch({ + owner: context.repo.owner, + repo: context.repo.repo, + workflow_id: 'rvs-nightly-docker-ubuntu24.04-tests.yml', + ref, + inputs, + }); + } catch (error) { + if (error.status === 422 && String(error.message).includes('Unexpected inputs')) { + core.warning('rvs-nightly-docker-ubuntu24.04-tests.yml on ref does not define trigger_event — dispatching without it.'); + await github.rest.actions.createWorkflowDispatch({ + owner: context.repo.owner, + repo: context.repo.repo, + workflow_id: 'rvs-nightly-docker-ubuntu24.04-tests.yml', + ref, + inputs: {}, + }); + } else { + throw error; + } + } + core.info(`Dispatched RVS Nightly Docker Tests Ubuntu 24.04 (trigger_event=${context.eventName}) on ref ${ref}`); + + - name: Dispatch RVS Release Tests + if: startsWith(github.ref_name, 'release/') + uses: actions/github-script@v7 + with: + script: | + const ref = context.ref.replace('refs/heads/', ''); + await github.rest.actions.createWorkflowDispatch({ + owner: context.repo.owner, + repo: context.repo.repo, + workflow_id: 'rvs-release-tests.yml', + ref, + inputs: {}, + }); + core.info(`Dispatched RVS Release Tests on ref ${ref}`); + + - name: Dispatch RVS PR Tests + if: >- + github.event_name == 'pull_request' && + github.event.pull_request.head.repo.full_name == github.repository + env: + GITHUB_REF_NAME: ${{ github.ref_name }} + uses: actions/github-script@v7 + with: + script: | + const prNumber = String(context.payload.pull_request.number); + const buildRunNumber = String(context.runNumber); + // Pass identifiers only — manylinux listing probe runs on the self-hosted + // runner in rvs-pr-tests.yml (lab CDN egress). Use PR head ref so the + // dispatched workflow includes pr_number / build_run_number inputs. + const ref = context.payload.pull_request.head.ref; + const buildRefName = process.env.GITHUB_REF_NAME || ''; + await github.rest.actions.createWorkflowDispatch({ + owner: context.repo.owner, + repo: context.repo.repo, + workflow_id: 'rvs-pr-tests.yml', + ref, + inputs: { + pr_number: prNumber, + build_run_number: buildRunNumber, + build_ref_name: buildRefName, + }, + }); + core.info( + `Dispatched RVS PR Tests for PR #${prNumber} (build run ${buildRunNumber}) on ref ${ref}`, + ); + + create-release: + name: Create Release Summary + needs: [build-ubuntu, build-centos] + runs-on: ${{ vars.RUNNER_LABEL_UTILITY || 'ubuntu-latest' }} + # Skip on schedule: matrix builds do not aggregate job outputs reliably + if: always() && !cancelled() && github.event_name != 'schedule' + + steps: + - name: Generate Build Report + run: | + cat > build-report.md << 'REPORT' + # ROCm Validation Suite - Build Report + + ## Build Information + - **Commit**: ${{ github.sha }} + - **Branch**: ${{ github.ref_name }} + - **Build Date**: $(date -u +'%Y-%m-%d %H:%M:%S UTC') + + ## Installation Instructions + + ### Ubuntu/Debian + ```bash + sudo dpkg -i amdrocm*-rvs_*.deb + ``` + + ### CentOS/RHEL + ```bash + sudo rpm -i --replacefiles --nodeps amdrocm*-rvs-*.rpm + ``` + + ### Relocatable TGZ (Any Linux Distribution) + ```bash + tar -xzf amdrocm*-rvs-*.tar.gz -C /opt/rocm/ + export PATH=/opt/rocm/rvs/bin:$PATH + export LD_LIBRARY_PATH=/opt/rocm/rvs/lib:$LD_LIBRARY_PATH + ``` + + ## Notes + - All packages are built with relocatable RPATH settings + - TGZ packages can be extracted to any location + - Install path: /opt/rocm/rvs + - ROCm SDK from TheRock nightly builds is used + REPORT + + - name: Add S3 upload links to build report + run: | + BUCKET="${{ needs.build-ubuntu.outputs.s3_bucket_ubuntu }}" + if [ -z "$BUCKET" ]; then + BUCKET="${{ needs.build-centos.outputs.s3_bucket_centos }}" + fi + if [ -z "$BUCKET" ]; then + echo "No S3 upload in this run, skipping S3 links section." + exit 0 + fi + echo "" >> build-report.md + echo "## S3 Upload Locations" >> build-report.md + echo "" >> build-report.md + echo "Packages were uploaded to the following S3 locations (AWS Console links):" >> build-report.md + echo "" >> build-report.md + PATHS_UBUNTU="${{ needs.build-ubuntu.outputs.s3_paths_ubuntu }}" + PATHS_CENTOS="${{ needs.build-centos.outputs.s3_paths_centos }}" + echo "$PATHS_UBUNTU" | sed 's/||/\n/g' | while IFS= read -r part; do + [ -z "$part" ] && continue + label="${part%%|*}"; prefix="${part#*|}" + enc=$(echo "$prefix" | sed 's|/|%2F|g')/ + echo "- [$label](https://s3.console.aws.amazon.com/s3/buckets/${BUCKET}?prefix=${enc})" >> build-report.md + done + echo "$PATHS_CENTOS" | sed 's/||/\n/g' | while IFS= read -r part; do + [ -z "$part" ] && continue + label="${part%%|*}"; prefix="${part#*|}" + enc=$(echo "$prefix" | sed 's|/|%2F|g')/ + echo "- [$label](https://s3.console.aws.amazon.com/s3/buckets/${BUCKET}?prefix=${enc})" >> build-report.md + done + echo "" >> build-report.md + cat build-report.md + + - name: Upload Build Report + uses: actions/upload-artifact@v4 + with: + name: build-report + path: build-report.md + retention-days: 90 diff --git a/.github/workflows/rvs-nightly-docker-manylinux_2_28-tests.yml b/.github/workflows/rvs-nightly-docker-manylinux_2_28-tests.yml new file mode 100644 index 000000000..c855f3160 --- /dev/null +++ b/.github/workflows/rvs-nightly-docker-manylinux_2_28-tests.yml @@ -0,0 +1,335 @@ +name: RVS Nightly Tests (Docker manylinux_2_28) + +# Same orchestration as rvs-nightly-docker-rhel8-tests.yml (workflow_dispatch, +# build-on-target, level 4 in docker), but the runtime image is +# quay.io/pypa/manylinux_2_28_x86_64 and ROCm is the TheRock multiarch SDK +# (therock-dist-linux-multiarch-.tar.gz), not yum. RVS is installed +# from the relocatable *-Linux.tar.gz (same as ubuntu docker nightly). +# +# Default image delivery: build-on-target (docker build on the GPU node). +# Set workflow input build_on_target=false to fall back to scp/registry delivery. +# +# Prerequisites: +# - Build host (self-hosted runner): curl/scp/ssh; HTTPS to tarball index and nightly.repo.amd.com +# - Target GPU node: docker, HTTPS to nightly.repo.amd.com, /dev/kfd, /dev/dri, SSH user in docker group +# - ROCm SDK in docker image: therock-dist-linux-multiarch-.tar.gz on manylinux_2_28 +# - Secrets: RVS_TARGET_NODE, RVS_TARGET_SSH_KEY (optional RVS_TARGET_USER, RVS_TARBALL_INDEX_URL) +# - Optional: vars.RVS_NIGHTLY_DOCKER_MANYLINUX_IMAGE, vars.RVS_DOCKER_REGISTRY, vars.ROCM_SDK_NIGHTLY_BASE_URL + +on: + workflow_dispatch: + inputs: + trigger_event: + description: 'Label for report only (e.g. schedule or push); optional' + required: false + default: '' + tarball_url: + description: 'Override tarball URL (default: latest from index — secrets.RVS_TARBALL_INDEX_URL)' + required: false + default: '' + target_node: + description: 'GPU target hostname/IP (default: secrets.RVS_TARGET_NODE)' + required: false + default: '' + target_user: + description: 'SSH user on target (default: secrets.RVS_TARGET_USER)' + required: false + default: '' + docker_image: + description: 'ROCm runtime image (default: vars.RVS_NIGHTLY_DOCKER_MANYLINUX_IMAGE or rvs-nightly-rocm-manylinux_2_28:latest)' + required: false + default: '' + build_docker_image: + description: 'Build rvs-nightly-rocm-manylinux_2_28 on the build host before scp/registry delivery (only when build_on_target is false)' + required: false + type: boolean + default: false + build_on_target: + description: 'Build the ROCm docker image on the GPU target (default; skips cross-host image transfer)' + required: false + type: boolean + default: true + docker_transfer_mode: + description: 'Image delivery — auto, scp, registry, or build-on-target (default auto)' + required: false + default: 'auto' + remote_work_dir: + description: 'Work dir on target (default: /tmp/rvs-nightly-docker-manylinux_2_28-)' + required: false + default: '' + fallback_latest_sdk: + description: 'If the exact SDK date is missing, fall back to latest same line, then same major, then newest nightly' + required: false + type: boolean + default: true + +permissions: + contents: read + +concurrency: + group: rvs-nightly-docker-manylinux_2_28-${{ github.workflow }} + cancel-in-progress: false + +env: + TARBALL_INDEX_URL: ${{ secrets.RVS_TARBALL_INDEX_URL }} + RVS_NIGHTLY_DOCKER_IMAGE: ${{ inputs.docker_image || vars.RVS_NIGHTLY_DOCKER_MANYLINUX_IMAGE || 'rvs-nightly-rocm-manylinux_2_28:latest' }} + RVS_DOCKER_BUILD_DIR: .github/docker/rvs-nightly-rocm-manylinux_2_28 + RVS_DOCKER_IMAGE_ARCHIVE: rvs-nightly-rocm-manylinux_2_28-image.tar.gz + RVS_DOCKER_ON_TARGET: 'true' + RVS_DOCKER_TRANSFER_MODE: ${{ inputs.docker_transfer_mode || vars.RVS_DOCKER_TRANSFER_MODE || 'auto' }} + RVS_DOCKER_BUILD_ON_TARGET: ${{ (inputs.build_on_target != false) && (vars.RVS_DOCKER_BUILD_ON_TARGET != 'false') && 'true' || 'false' }} + RVS_DOCKER_REGISTRY: ${{ vars.RVS_DOCKER_REGISTRY || '' }} + RVS_DOCKER_REGISTRY_USER: ${{ secrets.RVS_DOCKER_REGISTRY_USER }} + RVS_DOCKER_REGISTRY_PASSWORD: ${{ secrets.RVS_DOCKER_REGISTRY_PASSWORD }} + RVS_DOCKER_SKIP_IF_PRESENT: 'true' + RVS_DOCKER_SDK_FALLBACK_LATEST: ${{ (github.event.inputs.fallback_latest_sdk != 'false' && vars.RVS_DOCKER_SDK_FALLBACK_LATEST != 'false') && 'true' || 'false' }} + # ROCm SDK nightly host. Set vars.ROCM_SDK_NIGHTLY_BASE_URL (e.g. https://nightly.repo.amd.com/rocm/core/tarball). + # Optional vars.ROCM_SDK_NIGHTLY_INDEX_URL; if unset, BASE_URL is used for both listing and download. + ROCM_SDK_NIGHTLY_BASE_URL: ${{ vars.ROCM_SDK_NIGHTLY_BASE_URL || '' }} + ROCM_SDK_NIGHTLY_INDEX_URL: ${{ vars.ROCM_SDK_NIGHTLY_INDEX_URL || vars.ROCM_SDK_NIGHTLY_BASE_URL || '' }} + TARGET_ROCM_PATH: /opt/rocm/install + TARGET_NODE: ${{ inputs.target_node || secrets.RVS_TARGET_NODE }} + TARGET_USER: ${{ inputs.target_user || secrets.RVS_TARGET_USER }} + SSH_PRIVATE_KEY: ${{ secrets.RVS_TARGET_SSH_KEY }} + +jobs: + install-rvs-in-docker: + name: Ensure ROCm image (manylinux_2_28) on target and install RVS in docker + runs-on: ${{ vars.RVS_NIGHTLY_DOCKER_RUNNER_LABEL || vars.RVS_TEST_RUNNER_LABEL || 'self-hosted' }} + timeout-minutes: 180 + outputs: + tarball_url: ${{ steps.resolve.outputs.tarball_url }} + tarball_name: ${{ steps.resolve.outputs.tarball_name }} + remote_work_dir: ${{ steps.prepare.outputs.remote_work_dir }} + rocm_major: ${{ steps.prepare.outputs.rocm_major }} + install_dir: ${{ steps.prepare.outputs.install_dir }} + rvs_bin: ${{ steps.prepare.outputs.rvs_bin }} + env: + TARGET_ROCM_PATH: /opt/rocm/install + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + + - name: Resolve latest tarball URL + id: resolve + env: + INPUT_URL: ${{ inputs.tarball_url }} + run: | + set -euo pipefail + INDEX_URL="${TARBALL_INDEX_URL:-}" + if [ -z "${INPUT_URL:-}" ] && [ -z "${INDEX_URL}" ]; then + echo "::error::No tarball index URL (set RVS_TARBALL_INDEX_URL) and no tarball_url input." + exit 1 + fi + fetch_latest() { + local idx="$1" + local html raw name + html="$(curl -sSL --max-time 60 --retry 2 --retry-delay 2 "$idx" 2>/dev/null || true)" + raw="$(printf '%s' "$html" | grep -oE 'amdrocm[0-9]*-rvs-[0-9A-Za-z._\-]+-Linux\.tar\.gz' 2>/dev/null || true)" + name="$(printf '%s\n' "$raw" | sort -uV | tail -n 1 || true)" + [ -n "$name" ] || return 1 + printf '%s/%s\n' "${idx%/}" "$name" + } + if [ -n "${INPUT_URL:-}" ]; then + URL="$INPUT_URL" + else + URL="$(fetch_latest "$INDEX_URL")" + fi + [ -n "${URL:-}" ] || { echo "::error::Could not resolve tarball URL"; exit 1; } + NAME="$(basename "$URL" | tr -d '\r\n')" + if [[ "$NAME" != *-Linux.tar.gz ]]; then + echo "::error::Docker nightly tests require a *-Linux.tar.gz relocatable tarball; got: ${NAME}" >&2 + exit 1 + fi + if [[ ! "$NAME" =~ -r[0-9]{4}\.[0-9]{8}-Linux\.tar\.gz$ ]]; then + echo "::error::Tar tarball must include -rMMmm.yyyymmdd- (e.g. -r0715.20260724-Linux.tar.gz); got: ${NAME}" >&2 + exit 1 + fi + { + echo "tarball_url<> "$GITHUB_OUTPUT" + { + echo "TARBALL_URL<> "$GITHUB_ENV" + echo "Latest tarball : $NAME" + + - name: Validate configuration and derive paths + id: prepare + env: + TARBALL_NAME: ${{ steps.resolve.outputs.tarball_name }} + INPUT_REMOTE_WORK_DIR: ${{ inputs.remote_work_dir || format('/tmp/rvs-nightly-docker-manylinux_2_28-{0}', github.run_id) }} + VAR_REMOTE_WORK_DIR: ${{ vars.RVS_REMOTE_WORK_DIR }} + GITHUB_RUN_ID: ${{ github.run_id }} + run: | + chmod +x rvs_nightly_test.sh + if [ -z "${TARGET_NODE:-}" ]; then + echo "::error::No target node (set secrets.RVS_TARGET_NODE or workflow input target_node)." >&2 + exit 1 + fi + ./rvs_nightly_test.sh validate-config + + - name: Setup SSH to target node + env: + REMOTE_WORK_DIR: ${{ steps.prepare.outputs.remote_work_dir }} + run: ./rvs_nightly_test.sh setup-ssh + + - name: Export paths for remaining steps + env: + TARBALL_NAME: ${{ steps.resolve.outputs.tarball_name }} + run: | + { + echo "REMOTE_WORK_DIR=${{ steps.prepare.outputs.remote_work_dir }}" + echo "ROCM_MAJOR=${{ steps.prepare.outputs.rocm_major }}" + echo "INSTALL_DIR=${{ steps.prepare.outputs.install_dir }}" + echo "RVS_BIN=${{ steps.prepare.outputs.rvs_bin }}" + } >> "$GITHUB_ENV" + if [[ "$TARBALL_NAME" =~ -r([0-9]{2})([0-9]{2})\.([0-9]{8})-Linux\.tar\.gz$ ]]; then + echo "RVS_DOCKER_ROCM_VERSION=$((10#${BASH_REMATCH[1]})).$((10#${BASH_REMATCH[2]})).0a${BASH_REMATCH[3]}" >> "$GITHUB_ENV" + fi + + - name: Build ROCm docker image on build host + if: inputs.build_docker_image && inputs.build_on_target == false + env: + TARBALL_NAME: ${{ steps.resolve.outputs.tarball_name }} + run: | + chmod +x .github/docker/build-rocm-sdk-image.sh + fallback_args=() + if [ "${RVS_DOCKER_SDK_FALLBACK_LATEST}" = true ]; then + fallback_args=(--fallback-latest-sdk) + fi + ./.github/docker/build-rocm-sdk-image.sh --context .github/docker/rvs-nightly-rocm-manylinux_2_28 --from-tarball "$TARBALL_NAME" "${fallback_args[@]}" + + - name: Verify ROCm docker image on build host + if: inputs.build_docker_image && inputs.build_on_target == false + run: | + chmod +x rvs_nightly_docker.sh + ./rvs_nightly_docker.sh pull-image + + - name: Ensure ROCm docker image on GPU target + env: + TARBALL_NAME: ${{ steps.resolve.outputs.tarball_name }} + run: | + chmod +x rvs_nightly_docker.sh + ./rvs_nightly_docker.sh ensure-image-on-target + + - name: Verify ROCm in docker on target + run: ./rvs_nightly_docker.sh verify-rocm + + - name: Install RVS in docker on target + run: ./rvs_nightly_docker.sh install-rvs + + run-rvs-level-4-in-docker: + name: Run RVS level 4 in docker (manylinux_2_28) on target + needs: [install-rvs-in-docker] + runs-on: ${{ vars.RVS_NIGHTLY_DOCKER_RUNNER_LABEL || vars.RVS_TEST_RUNNER_LABEL || 'self-hosted' }} + timeout-minutes: 180 + outputs: + rc: ${{ steps.level4.outputs.rc }} + start: ${{ steps.level4.outputs.start }} + end: ${{ steps.level4.outputs.end }} + rvs_version: ${{ steps.versions.outputs.rvs_version }} + target_rocm_version: ${{ steps.versions.outputs.target_rocm_version }} + env: + TARGET_ROCM_PATH: /opt/rocm/install + TARGET_NODE: ${{ inputs.target_node || secrets.RVS_TARGET_NODE }} + TARGET_USER: ${{ inputs.target_user || secrets.RVS_TARGET_USER }} + SSH_PRIVATE_KEY: ${{ secrets.RVS_TARGET_SSH_KEY }} + RVS_DOCKER_ON_TARGET: 'true' + RVS_DOCKER_BUILD_DIR: .github/docker/rvs-nightly-rocm-manylinux_2_28 + RVS_DOCKER_IMAGE_ARCHIVE: rvs-nightly-rocm-manylinux_2_28-image.tar.gz + REMOTE_WORK_DIR: ${{ needs.install-rvs-in-docker.outputs.remote_work_dir }} + ROCM_MAJOR: ${{ needs.install-rvs-in-docker.outputs.rocm_major }} + INSTALL_DIR: ${{ needs.install-rvs-in-docker.outputs.install_dir }} + RVS_BIN: ${{ needs.install-rvs-in-docker.outputs.rvs_bin }} + RVS_NIGHTLY_DOCKER_IMAGE: ${{ inputs.docker_image || vars.RVS_NIGHTLY_DOCKER_MANYLINUX_IMAGE || 'rvs-nightly-rocm-manylinux_2_28:latest' }} + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + + - name: Setup SSH to target node + env: + REMOTE_WORK_DIR: ${{ needs.install-rvs-in-docker.outputs.remote_work_dir }} + run: | + chmod +x rvs_nightly_test.sh rvs_nightly_docker.sh + ./rvs_nightly_test.sh setup-ssh + + - name: Run RVS level 4 in docker on target + id: level4 + run: ./rvs_nightly_docker.sh run-level4 + + - name: Capture RVS and ROCm versions from docker on target + id: versions + if: always() + run: ./rvs_nightly_docker.sh capture-versions + + - name: Collect logs from target + if: always() + run: ./rvs_nightly_test.sh collect-logs + + - name: Upload test logs + if: always() + uses: actions/upload-artifact@v4 + with: + name: rvs-nightly-docker-manylinux_2_28-logs-${{ github.run_id }} + path: ./reports/ + retention-days: 30 + if-no-files-found: warn + + create-test-report: + name: Create Test Report + needs: [install-rvs-in-docker, run-rvs-level-4-in-docker] + runs-on: ${{ vars.RUNNER_LABEL_UTILITY || 'ubuntu-latest' }} + if: always() && !cancelled() && needs.install-rvs-in-docker.result == 'success' && needs.run-rvs-level-4-in-docker.result != 'skipped' + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + + - name: Download test logs + uses: actions/download-artifact@v4 + with: + name: rvs-nightly-docker-manylinux_2_28-logs-${{ github.run_id }} + path: ./reports + continue-on-error: true + + - name: Build test report + id: report + env: + TARBALL_URL: ${{ needs.install-rvs-in-docker.outputs.tarball_url }} + TARBALL_NAME: ${{ needs.install-rvs-in-docker.outputs.tarball_name }} + TARGET_ROCM_PATH: /opt/rocm/install + REMOTE_WORK_DIR: ${{ needs.install-rvs-in-docker.outputs.remote_work_dir }} + RVS_BIN: ${{ needs.install-rvs-in-docker.outputs.rvs_bin }} + RVS_LEVEL4_RC: ${{ needs.run-rvs-level-4-in-docker.outputs.rc }} + RVS_LEVEL4_START: ${{ needs.run-rvs-level-4-in-docker.outputs.start }} + RVS_LEVEL4_END: ${{ needs.run-rvs-level-4-in-docker.outputs.end }} + RVS_VERSION: ${{ needs.run-rvs-level-4-in-docker.outputs.rvs_version }} + TARGET_ROCM_VERSION: ${{ needs.run-rvs-level-4-in-docker.outputs.target_rocm_version }} + GITHUB_RUN_ID: ${{ github.run_id }} + GITHUB_SERVER_URL: ${{ github.server_url }} + GITHUB_REPOSITORY: ${{ github.repository }} + GITHUB_EVENT_NAME: ${{ inputs.trigger_event || github.event_name }} + run: | + chmod +x rvs_nightly_test.sh + mkdir -p ./reports + ./rvs_nightly_test.sh build-report + + - name: Upload report and logs + if: always() + uses: actions/upload-artifact@v4 + with: + name: rvs-nightly-docker-manylinux_2_28-report-${{ github.run_id }} + path: ./reports/ + retention-days: 30 + if-no-files-found: warn + + - name: Fail the workflow if level 4 failed + if: steps.report.outputs.overall == 'FAIL' || needs.run-rvs-level-4-in-docker.result == 'failure' + run: | + echo "::error::RVS docker (manylinux_2_28) test failure — level4 rc=${{ needs.run-rvs-level-4-in-docker.outputs.rc }}" + exit 1 diff --git a/.github/workflows/rvs-nightly-docker-rhel8-tests.yml b/.github/workflows/rvs-nightly-docker-rhel8-tests.yml new file mode 100644 index 000000000..18564f611 --- /dev/null +++ b/.github/workflows/rvs-nightly-docker-rhel8-tests.yml @@ -0,0 +1,317 @@ +name: RVS Nightly Tests (Docker RHEL 8) + +# Same orchestration as rvs-nightly-docker-ubuntu22.04-tests.yml, but the runtime +# image is rockylinux:8 and ROCm + RVS are installed from AMD nightly yum repos +# (dnf), not from relocatable tarballs. The stable 10.0.0 repo URLs / package +# names are examples only — this workflow always tracks nightly. +# +# Install sequence inside the image: +# dnf update → dnf install sudo wget → nightly ROCm yum (amdrocmX.Y multiarch, or amdrocmX.Y-gfx* if gpu_target is set) → +# extras yum for RVS (nightly if published, else stable extras) via rpm --nodeps +# so the unversioned amdrocm-* metas do not pull every GPU ISA. +# +# Default image delivery: build-on-target (docker build on the GPU node). +# Set workflow input build_on_target=false to fall back to scp/registry delivery. +# +# Prerequisites: +# - Build host (self-hosted runner): curl/scp/ssh; HTTPS to nightly.repo.amd.com +# - Target GPU node: docker, /dev/kfd, /dev/dri, SSH user in docker group +# - Secrets: RVS_TARGET_NODE, RVS_TARGET_SSH_KEY (optional RVS_TARGET_USER) +# - Optional vars: RVS_NIGHTLY_RHEL8_ROCM_REPO_INDEX, RVS_NIGHTLY_RHEL8_RVS_REPO_BASEURL, +# RVS_NIGHTLY_DOCKER_RHEL8_IMAGE, RVS_NIGHTLY_RHEL8_GPU_TARGET + +on: + workflow_dispatch: + inputs: + trigger_event: + description: 'Label for report only (e.g. schedule or push); optional' + required: false + default: '' + target_node: + description: 'GPU target hostname/IP (default: secrets.RVS_TARGET_NODE)' + required: false + default: '' + target_user: + description: 'SSH user on target (default: secrets.RVS_TARGET_USER)' + required: false + default: '' + docker_image: + description: 'ROCm runtime image (default: vars.RVS_NIGHTLY_DOCKER_RHEL8_IMAGE or rvs-nightly-rocm-rhel8:latest)' + required: false + default: '' + gpu_target: + description: 'GPU ISA package suffix (default: multiarch → amdrocmX.Y; or gfx942, gfx950, …)' + required: false + default: '' + rocm_repo_baseurl: + description: 'Override nightly ROCm yum baseurl (default: latest snapshot under nightly.repo.amd.com .../rhel8//x86_64)' + required: false + default: '' + rvs_repo_baseurl: + description: 'Override RVS yum baseurl (default: nightly extras if published, else stable extras rhel8 x86_64)' + required: false + default: '' + build_docker_image: + description: 'Build rvs-nightly-rocm-rhel8 on the build host before scp/registry delivery (only when build_on_target is false)' + required: false + type: boolean + default: false + build_on_target: + description: 'Build the ROCm docker image on the GPU target (default; skips cross-host image transfer)' + required: false + type: boolean + default: true + docker_transfer_mode: + description: 'Image delivery — auto, scp, registry, or build-on-target (default auto)' + required: false + default: 'auto' + remote_work_dir: + description: 'Work dir on target (default: /tmp/rvs-nightly-docker-rhel8-)' + required: false + default: '' + +permissions: + contents: read + +concurrency: + group: rvs-nightly-docker-rhel8-${{ github.workflow }} + cancel-in-progress: false + +env: + RVS_NIGHTLY_DOCKER_IMAGE: ${{ inputs.docker_image || vars.RVS_NIGHTLY_DOCKER_RHEL8_IMAGE || 'rvs-nightly-rocm-rhel8:latest' }} + RVS_DOCKER_BUILD_DIR: .github/docker/rvs-nightly-rocm-rhel8 + RVS_DOCKER_IMAGE_ARCHIVE: rvs-nightly-rocm-rhel8-image.tar.gz + RVS_DOCKER_ON_TARGET: 'true' + RVS_DOCKER_TRANSFER_MODE: ${{ inputs.docker_transfer_mode || vars.RVS_DOCKER_TRANSFER_MODE || 'auto' }} + RVS_DOCKER_BUILD_ON_TARGET: ${{ (inputs.build_on_target != false) && (vars.RVS_DOCKER_BUILD_ON_TARGET != 'false') && 'true' || 'false' }} + RVS_DOCKER_REGISTRY: ${{ vars.RVS_DOCKER_REGISTRY || '' }} + RVS_DOCKER_REGISTRY_USER: ${{ secrets.RVS_DOCKER_REGISTRY_USER }} + RVS_DOCKER_REGISTRY_PASSWORD: ${{ secrets.RVS_DOCKER_REGISTRY_PASSWORD }} + RVS_DOCKER_SKIP_IF_PRESENT: 'true' + RVS_DOCKER_RVS_IN_IMAGE: 'true' + GPU_TARGET: ${{ inputs.gpu_target || vars.RVS_NIGHTLY_RHEL8_GPU_TARGET || 'multiarch' }} + TARGET_ROCM_PATH: /opt/rocm + TARGET_NODE: ${{ inputs.target_node || secrets.RVS_TARGET_NODE }} + TARGET_USER: ${{ inputs.target_user || secrets.RVS_TARGET_USER }} + SSH_PRIVATE_KEY: ${{ secrets.RVS_TARGET_SSH_KEY }} + +jobs: + install-rvs-in-docker: + name: Ensure ROCm image (RHEL 8) on target and install RVS in docker + runs-on: ${{ vars.RVS_NIGHTLY_DOCKER_RUNNER_LABEL || vars.RVS_TEST_RUNNER_LABEL || 'self-hosted' }} + timeout-minutes: 180 + outputs: + tarball_url: ${{ steps.resolve.outputs.tarball_url }} + tarball_name: ${{ steps.resolve.outputs.tarball_name }} + remote_work_dir: ${{ steps.prepare.outputs.remote_work_dir }} + rocm_major: ${{ steps.prepare.outputs.rocm_major }} + install_dir: ${{ steps.prepare.outputs.install_dir }} + rvs_bin: ${{ steps.prepare.outputs.rvs_bin }} + env: + TARGET_ROCM_PATH: /opt/rocm + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + + - name: Resolve latest nightly ROCm and RVS yum repos + id: resolve + env: + INPUT_ROCM_REPO: ${{ inputs.rocm_repo_baseurl }} + INPUT_RVS_REPO: ${{ inputs.rvs_repo_baseurl || vars.RVS_NIGHTLY_RHEL8_RVS_REPO_BASEURL }} + RVS_NIGHTLY_RHEL8_ROCM_REPO_INDEX: ${{ vars.RVS_NIGHTLY_RHEL8_ROCM_REPO_INDEX }} + run: | + set -euo pipefail + chmod +x .github/docker/build-rocm-sdk-image.sh + resolve_args=(--resolve-only --channel nightly --gpu-target "${GPU_TARGET}") + if [ -n "${INPUT_ROCM_REPO:-}" ]; then + resolve_args+=(--rocm-repo "${INPUT_ROCM_REPO}") + fi + if [ -n "${INPUT_RVS_REPO:-}" ]; then + resolve_args+=(--rvs-repo "${INPUT_RVS_REPO}") + fi + ./.github/docker/build-rocm-sdk-image.sh --context .github/docker/rvs-nightly-rocm-rhel8 "${resolve_args[@]}" + + - name: Export resolved repo paths + env: + TARBALL_NAME: ${{ steps.resolve.outputs.tarball_name }} + TARBALL_URL: ${{ steps.resolve.outputs.tarball_url }} + run: | + { + echo "TARBALL_NAME=${TARBALL_NAME}" + echo "TARBALL_URL=${TARBALL_URL}" + echo "ROCM_REPO_BASEURL=${{ steps.resolve.outputs.rocm_repo_baseurl }}" + echo "RVS_REPO_BASEURL=${{ steps.resolve.outputs.rvs_repo_baseurl }}" + echo "ROCM_PACKAGE=${{ steps.resolve.outputs.rocm_package }}" + echo "RVS_PACKAGE=${{ steps.resolve.outputs.rvs_package }}" + echo "RVS_DOCKER_ROCM_VERSION=${{ steps.resolve.outputs.rocm_version }}" + echo "ROCM_MAJOR=${{ steps.resolve.outputs.rocm_major }}" + } >> "$GITHUB_ENV" + + - name: Validate configuration and derive paths + id: prepare + env: + TARBALL_NAME: ${{ steps.resolve.outputs.tarball_name }} + INPUT_REMOTE_WORK_DIR: ${{ inputs.remote_work_dir || format('/tmp/rvs-nightly-docker-rhel8-{0}', github.run_id) }} + VAR_REMOTE_WORK_DIR: ${{ vars.RVS_REMOTE_WORK_DIR }} + GITHUB_RUN_ID: ${{ github.run_id }} + run: | + chmod +x rvs_nightly_test.sh + if [ -z "${TARGET_NODE:-}" ]; then + echo "::error::No target node (set secrets.RVS_TARGET_NODE or workflow input target_node)." >&2 + exit 1 + fi + ./rvs_nightly_test.sh validate-config + + - name: Setup SSH to target node + env: + REMOTE_WORK_DIR: ${{ steps.prepare.outputs.remote_work_dir }} + run: ./rvs_nightly_test.sh setup-ssh + + - name: Export paths for remaining steps + run: | + { + echo "REMOTE_WORK_DIR=${{ steps.prepare.outputs.remote_work_dir }}" + echo "ROCM_MAJOR=${{ steps.prepare.outputs.rocm_major }}" + echo "INSTALL_DIR=${{ steps.prepare.outputs.install_dir }}" + echo "RVS_BIN=${{ steps.prepare.outputs.rvs_bin }}" + } >> "$GITHUB_ENV" + + - name: Build ROCm docker image on build host + if: inputs.build_docker_image && inputs.build_on_target == false + run: | + chmod +x .github/docker/build-rocm-sdk-image.sh + ./.github/docker/build-rocm-sdk-image.sh --context .github/docker/rvs-nightly-rocm-rhel8 \ + --channel nightly \ + --gpu-target "${GPU_TARGET}" \ + --rocm-repo "${ROCM_REPO_BASEURL}" \ + --rvs-repo "${RVS_REPO_BASEURL}" \ + --rocm-package "${ROCM_PACKAGE}" \ + --rvs-package "${RVS_PACKAGE}" + + - name: Verify ROCm docker image on build host + if: inputs.build_docker_image && inputs.build_on_target == false + run: | + chmod +x rvs_nightly_docker.sh + ./rvs_nightly_docker.sh pull-image + + - name: Ensure ROCm docker image on GPU target + run: | + chmod +x rvs_nightly_docker.sh + ./rvs_nightly_docker.sh ensure-image-on-target + + - name: Verify ROCm in docker on target (rocminfo) + run: ./rvs_nightly_docker.sh verify-rocm + + - name: Verify RVS in docker on target + run: ./rvs_nightly_docker.sh install-rvs + + run-rvs-level-4-in-docker: + name: Run RVS level 4 in docker (RHEL 8) on target + needs: [install-rvs-in-docker] + runs-on: ${{ vars.RVS_NIGHTLY_DOCKER_RUNNER_LABEL || vars.RVS_TEST_RUNNER_LABEL || 'self-hosted' }} + timeout-minutes: 180 + outputs: + rc: ${{ steps.level4.outputs.rc }} + start: ${{ steps.level4.outputs.start }} + end: ${{ steps.level4.outputs.end }} + rvs_version: ${{ steps.versions.outputs.rvs_version }} + target_rocm_version: ${{ steps.versions.outputs.target_rocm_version }} + env: + TARGET_ROCM_PATH: /opt/rocm + TARGET_NODE: ${{ inputs.target_node || secrets.RVS_TARGET_NODE }} + TARGET_USER: ${{ inputs.target_user || secrets.RVS_TARGET_USER }} + SSH_PRIVATE_KEY: ${{ secrets.RVS_TARGET_SSH_KEY }} + RVS_DOCKER_ON_TARGET: 'true' + RVS_DOCKER_RVS_IN_IMAGE: 'true' + RVS_DOCKER_BUILD_DIR: .github/docker/rvs-nightly-rocm-rhel8 + RVS_DOCKER_IMAGE_ARCHIVE: rvs-nightly-rocm-rhel8-image.tar.gz + REMOTE_WORK_DIR: ${{ needs.install-rvs-in-docker.outputs.remote_work_dir }} + ROCM_MAJOR: ${{ needs.install-rvs-in-docker.outputs.rocm_major }} + INSTALL_DIR: ${{ needs.install-rvs-in-docker.outputs.install_dir }} + RVS_BIN: ${{ needs.install-rvs-in-docker.outputs.rvs_bin }} + RVS_NIGHTLY_DOCKER_IMAGE: ${{ inputs.docker_image || vars.RVS_NIGHTLY_DOCKER_RHEL8_IMAGE || 'rvs-nightly-rocm-rhel8:latest' }} + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + + - name: Setup SSH to target node + env: + REMOTE_WORK_DIR: ${{ needs.install-rvs-in-docker.outputs.remote_work_dir }} + run: | + chmod +x rvs_nightly_test.sh rvs_nightly_docker.sh + ./rvs_nightly_test.sh setup-ssh + + - name: Run RVS level 4 in docker on target + id: level4 + run: ./rvs_nightly_docker.sh run-level4 + + - name: Capture RVS and ROCm versions from docker on target + id: versions + if: always() + run: ./rvs_nightly_docker.sh capture-versions + + - name: Collect logs from target + if: always() + run: ./rvs_nightly_test.sh collect-logs + + - name: Upload test logs + if: always() + uses: actions/upload-artifact@v4 + with: + name: rvs-nightly-docker-rhel8-logs-${{ github.run_id }} + path: ./reports/ + retention-days: 30 + if-no-files-found: warn + + create-test-report: + name: Create Test Report + needs: [install-rvs-in-docker, run-rvs-level-4-in-docker] + runs-on: ${{ vars.RUNNER_LABEL_UTILITY || 'ubuntu-latest' }} + if: always() && !cancelled() && needs.install-rvs-in-docker.result == 'success' && needs.run-rvs-level-4-in-docker.result != 'skipped' + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + + - name: Download test logs + uses: actions/download-artifact@v4 + with: + name: rvs-nightly-docker-rhel8-logs-${{ github.run_id }} + path: ./reports + continue-on-error: true + + - name: Build test report + id: report + env: + TARBALL_URL: ${{ needs.install-rvs-in-docker.outputs.tarball_url }} + TARBALL_NAME: ${{ needs.install-rvs-in-docker.outputs.tarball_name }} + TARGET_ROCM_PATH: /opt/rocm + REMOTE_WORK_DIR: ${{ needs.install-rvs-in-docker.outputs.remote_work_dir }} + RVS_BIN: ${{ needs.install-rvs-in-docker.outputs.rvs_bin }} + RVS_LEVEL4_RC: ${{ needs.run-rvs-level-4-in-docker.outputs.rc }} + RVS_LEVEL4_START: ${{ needs.run-rvs-level-4-in-docker.outputs.start }} + RVS_LEVEL4_END: ${{ needs.run-rvs-level-4-in-docker.outputs.end }} + RVS_VERSION: ${{ needs.run-rvs-level-4-in-docker.outputs.rvs_version }} + TARGET_ROCM_VERSION: ${{ needs.run-rvs-level-4-in-docker.outputs.target_rocm_version }} + GITHUB_RUN_ID: ${{ github.run_id }} + GITHUB_SERVER_URL: ${{ github.server_url }} + GITHUB_REPOSITORY: ${{ github.repository }} + GITHUB_EVENT_NAME: ${{ inputs.trigger_event || github.event_name }} + run: | + chmod +x rvs_nightly_test.sh + mkdir -p ./reports + ./rvs_nightly_test.sh build-report + + - name: Upload report and logs + if: always() + uses: actions/upload-artifact@v4 + with: + name: rvs-nightly-docker-rhel8-report-${{ github.run_id }} + path: ./reports/ + retention-days: 30 + if-no-files-found: warn + + - name: Fail the workflow if level 4 failed + if: steps.report.outputs.overall == 'FAIL' || needs.run-rvs-level-4-in-docker.result == 'failure' + run: | + echo "::error::RVS docker (RHEL 8) test failure — level4 rc=${{ needs.run-rvs-level-4-in-docker.outputs.rc }}" + exit 1 diff --git a/.github/workflows/rvs-nightly-docker-rhel9-tests.yml b/.github/workflows/rvs-nightly-docker-rhel9-tests.yml new file mode 100644 index 000000000..76a181c07 --- /dev/null +++ b/.github/workflows/rvs-nightly-docker-rhel9-tests.yml @@ -0,0 +1,317 @@ +name: RVS Nightly Tests (Docker RHEL 9) + +# Same orchestration as rvs-nightly-docker-ubuntu22.04-tests.yml, but the runtime +# image is redhat/ubi9 and ROCm + RVS are installed from AMD nightly yum repos +# (dnf), not from relocatable tarballs. The stable 10.0.0 repo URLs / package +# names are examples only — this workflow always tracks nightly. +# +# Install sequence inside the image: +# dnf update → dnf install sudo wget → nightly ROCm yum (amdrocmX.Y multiarch, or amdrocmX.Y-gfx* if gpu_target is set) → +# extras yum for RVS (nightly if published, else stable extras) via rpm --nodeps +# so the unversioned amdrocm-* metas do not pull every GPU ISA. +# +# Default image delivery: build-on-target (docker build on the GPU node). +# Set workflow input build_on_target=false to fall back to scp/registry delivery. +# +# Prerequisites: +# - Build host (self-hosted runner): curl/scp/ssh; HTTPS to nightly.repo.amd.com +# - Target GPU node: docker, /dev/kfd, /dev/dri, SSH user in docker group +# - Secrets: RVS_TARGET_NODE, RVS_TARGET_SSH_KEY (optional RVS_TARGET_USER) +# - Optional vars: RVS_NIGHTLY_RHEL9_ROCM_REPO_INDEX, RVS_NIGHTLY_RHEL9_RVS_REPO_BASEURL, +# RVS_NIGHTLY_DOCKER_RHEL9_IMAGE, RVS_NIGHTLY_RHEL9_GPU_TARGET + +on: + workflow_dispatch: + inputs: + trigger_event: + description: 'Label for report only (e.g. schedule or push); optional' + required: false + default: '' + target_node: + description: 'GPU target hostname/IP (default: secrets.RVS_TARGET_NODE)' + required: false + default: '' + target_user: + description: 'SSH user on target (default: secrets.RVS_TARGET_USER)' + required: false + default: '' + docker_image: + description: 'ROCm runtime image (default: vars.RVS_NIGHTLY_DOCKER_RHEL9_IMAGE or rvs-nightly-rocm-rhel9:latest)' + required: false + default: '' + gpu_target: + description: 'GPU ISA package suffix (default: multiarch → amdrocmX.Y; or gfx942, gfx950, …)' + required: false + default: '' + rocm_repo_baseurl: + description: 'Override nightly ROCm yum baseurl (default: latest snapshot under nightly.repo.amd.com .../rhel9//x86_64)' + required: false + default: '' + rvs_repo_baseurl: + description: 'Override RVS yum baseurl (default: nightly extras if published, else stable extras rhel9 x86_64)' + required: false + default: '' + build_docker_image: + description: 'Build rvs-nightly-rocm-rhel9 on the build host before scp/registry delivery (only when build_on_target is false)' + required: false + type: boolean + default: false + build_on_target: + description: 'Build the ROCm docker image on the GPU target (default; skips cross-host image transfer)' + required: false + type: boolean + default: true + docker_transfer_mode: + description: 'Image delivery — auto, scp, registry, or build-on-target (default auto)' + required: false + default: 'auto' + remote_work_dir: + description: 'Work dir on target (default: /tmp/rvs-nightly-docker-rhel9-)' + required: false + default: '' + +permissions: + contents: read + +concurrency: + group: rvs-nightly-docker-rhel9-${{ github.workflow }} + cancel-in-progress: false + +env: + RVS_NIGHTLY_DOCKER_IMAGE: ${{ inputs.docker_image || vars.RVS_NIGHTLY_DOCKER_RHEL9_IMAGE || 'rvs-nightly-rocm-rhel9:latest' }} + RVS_DOCKER_BUILD_DIR: .github/docker/rvs-nightly-rocm-rhel9 + RVS_DOCKER_IMAGE_ARCHIVE: rvs-nightly-rocm-rhel9-image.tar.gz + RVS_DOCKER_ON_TARGET: 'true' + RVS_DOCKER_TRANSFER_MODE: ${{ inputs.docker_transfer_mode || vars.RVS_DOCKER_TRANSFER_MODE || 'auto' }} + RVS_DOCKER_BUILD_ON_TARGET: ${{ (inputs.build_on_target != false) && (vars.RVS_DOCKER_BUILD_ON_TARGET != 'false') && 'true' || 'false' }} + RVS_DOCKER_REGISTRY: ${{ vars.RVS_DOCKER_REGISTRY || '' }} + RVS_DOCKER_REGISTRY_USER: ${{ secrets.RVS_DOCKER_REGISTRY_USER }} + RVS_DOCKER_REGISTRY_PASSWORD: ${{ secrets.RVS_DOCKER_REGISTRY_PASSWORD }} + RVS_DOCKER_SKIP_IF_PRESENT: 'true' + RVS_DOCKER_RVS_IN_IMAGE: 'true' + GPU_TARGET: ${{ inputs.gpu_target || vars.RVS_NIGHTLY_RHEL9_GPU_TARGET || 'multiarch' }} + TARGET_ROCM_PATH: /opt/rocm + TARGET_NODE: ${{ inputs.target_node || secrets.RVS_TARGET_NODE }} + TARGET_USER: ${{ inputs.target_user || secrets.RVS_TARGET_USER }} + SSH_PRIVATE_KEY: ${{ secrets.RVS_TARGET_SSH_KEY }} + +jobs: + install-rvs-in-docker: + name: Ensure ROCm image (RHEL 9) on target and install RVS in docker + runs-on: ${{ vars.RVS_NIGHTLY_DOCKER_RUNNER_LABEL || vars.RVS_TEST_RUNNER_LABEL || 'self-hosted' }} + timeout-minutes: 180 + outputs: + tarball_url: ${{ steps.resolve.outputs.tarball_url }} + tarball_name: ${{ steps.resolve.outputs.tarball_name }} + remote_work_dir: ${{ steps.prepare.outputs.remote_work_dir }} + rocm_major: ${{ steps.prepare.outputs.rocm_major }} + install_dir: ${{ steps.prepare.outputs.install_dir }} + rvs_bin: ${{ steps.prepare.outputs.rvs_bin }} + env: + TARGET_ROCM_PATH: /opt/rocm + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + + - name: Resolve latest nightly ROCm and RVS yum repos + id: resolve + env: + INPUT_ROCM_REPO: ${{ inputs.rocm_repo_baseurl }} + INPUT_RVS_REPO: ${{ inputs.rvs_repo_baseurl || vars.RVS_NIGHTLY_RHEL9_RVS_REPO_BASEURL }} + RVS_NIGHTLY_RHEL9_ROCM_REPO_INDEX: ${{ vars.RVS_NIGHTLY_RHEL9_ROCM_REPO_INDEX }} + run: | + set -euo pipefail + chmod +x .github/docker/build-rocm-sdk-image.sh + resolve_args=(--resolve-only --channel nightly --gpu-target "${GPU_TARGET}") + if [ -n "${INPUT_ROCM_REPO:-}" ]; then + resolve_args+=(--rocm-repo "${INPUT_ROCM_REPO}") + fi + if [ -n "${INPUT_RVS_REPO:-}" ]; then + resolve_args+=(--rvs-repo "${INPUT_RVS_REPO}") + fi + ./.github/docker/build-rocm-sdk-image.sh --context .github/docker/rvs-nightly-rocm-rhel9 "${resolve_args[@]}" + + - name: Export resolved repo paths + env: + TARBALL_NAME: ${{ steps.resolve.outputs.tarball_name }} + TARBALL_URL: ${{ steps.resolve.outputs.tarball_url }} + run: | + { + echo "TARBALL_NAME=${TARBALL_NAME}" + echo "TARBALL_URL=${TARBALL_URL}" + echo "ROCM_REPO_BASEURL=${{ steps.resolve.outputs.rocm_repo_baseurl }}" + echo "RVS_REPO_BASEURL=${{ steps.resolve.outputs.rvs_repo_baseurl }}" + echo "ROCM_PACKAGE=${{ steps.resolve.outputs.rocm_package }}" + echo "RVS_PACKAGE=${{ steps.resolve.outputs.rvs_package }}" + echo "RVS_DOCKER_ROCM_VERSION=${{ steps.resolve.outputs.rocm_version }}" + echo "ROCM_MAJOR=${{ steps.resolve.outputs.rocm_major }}" + } >> "$GITHUB_ENV" + + - name: Validate configuration and derive paths + id: prepare + env: + TARBALL_NAME: ${{ steps.resolve.outputs.tarball_name }} + INPUT_REMOTE_WORK_DIR: ${{ inputs.remote_work_dir || format('/tmp/rvs-nightly-docker-rhel9-{0}', github.run_id) }} + VAR_REMOTE_WORK_DIR: ${{ vars.RVS_REMOTE_WORK_DIR }} + GITHUB_RUN_ID: ${{ github.run_id }} + run: | + chmod +x rvs_nightly_test.sh + if [ -z "${TARGET_NODE:-}" ]; then + echo "::error::No target node (set secrets.RVS_TARGET_NODE or workflow input target_node)." >&2 + exit 1 + fi + ./rvs_nightly_test.sh validate-config + + - name: Setup SSH to target node + env: + REMOTE_WORK_DIR: ${{ steps.prepare.outputs.remote_work_dir }} + run: ./rvs_nightly_test.sh setup-ssh + + - name: Export paths for remaining steps + run: | + { + echo "REMOTE_WORK_DIR=${{ steps.prepare.outputs.remote_work_dir }}" + echo "ROCM_MAJOR=${{ steps.prepare.outputs.rocm_major }}" + echo "INSTALL_DIR=${{ steps.prepare.outputs.install_dir }}" + echo "RVS_BIN=${{ steps.prepare.outputs.rvs_bin }}" + } >> "$GITHUB_ENV" + + - name: Build ROCm docker image on build host + if: inputs.build_docker_image && inputs.build_on_target == false + run: | + chmod +x .github/docker/build-rocm-sdk-image.sh + ./.github/docker/build-rocm-sdk-image.sh --context .github/docker/rvs-nightly-rocm-rhel9 \ + --channel nightly \ + --gpu-target "${GPU_TARGET}" \ + --rocm-repo "${ROCM_REPO_BASEURL}" \ + --rvs-repo "${RVS_REPO_BASEURL}" \ + --rocm-package "${ROCM_PACKAGE}" \ + --rvs-package "${RVS_PACKAGE}" + + - name: Verify ROCm docker image on build host + if: inputs.build_docker_image && inputs.build_on_target == false + run: | + chmod +x rvs_nightly_docker.sh + ./rvs_nightly_docker.sh pull-image + + - name: Ensure ROCm docker image on GPU target + run: | + chmod +x rvs_nightly_docker.sh + ./rvs_nightly_docker.sh ensure-image-on-target + + - name: Verify ROCm in docker on target (rocminfo) + run: ./rvs_nightly_docker.sh verify-rocm + + - name: Verify RVS in docker on target + run: ./rvs_nightly_docker.sh install-rvs + + run-rvs-level-4-in-docker: + name: Run RVS level 4 in docker (RHEL 9) on target + needs: [install-rvs-in-docker] + runs-on: ${{ vars.RVS_NIGHTLY_DOCKER_RUNNER_LABEL || vars.RVS_TEST_RUNNER_LABEL || 'self-hosted' }} + timeout-minutes: 180 + outputs: + rc: ${{ steps.level4.outputs.rc }} + start: ${{ steps.level4.outputs.start }} + end: ${{ steps.level4.outputs.end }} + rvs_version: ${{ steps.versions.outputs.rvs_version }} + target_rocm_version: ${{ steps.versions.outputs.target_rocm_version }} + env: + TARGET_ROCM_PATH: /opt/rocm + TARGET_NODE: ${{ inputs.target_node || secrets.RVS_TARGET_NODE }} + TARGET_USER: ${{ inputs.target_user || secrets.RVS_TARGET_USER }} + SSH_PRIVATE_KEY: ${{ secrets.RVS_TARGET_SSH_KEY }} + RVS_DOCKER_ON_TARGET: 'true' + RVS_DOCKER_RVS_IN_IMAGE: 'true' + RVS_DOCKER_BUILD_DIR: .github/docker/rvs-nightly-rocm-rhel9 + RVS_DOCKER_IMAGE_ARCHIVE: rvs-nightly-rocm-rhel9-image.tar.gz + REMOTE_WORK_DIR: ${{ needs.install-rvs-in-docker.outputs.remote_work_dir }} + ROCM_MAJOR: ${{ needs.install-rvs-in-docker.outputs.rocm_major }} + INSTALL_DIR: ${{ needs.install-rvs-in-docker.outputs.install_dir }} + RVS_BIN: ${{ needs.install-rvs-in-docker.outputs.rvs_bin }} + RVS_NIGHTLY_DOCKER_IMAGE: ${{ inputs.docker_image || vars.RVS_NIGHTLY_DOCKER_RHEL9_IMAGE || 'rvs-nightly-rocm-rhel9:latest' }} + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + + - name: Setup SSH to target node + env: + REMOTE_WORK_DIR: ${{ needs.install-rvs-in-docker.outputs.remote_work_dir }} + run: | + chmod +x rvs_nightly_test.sh rvs_nightly_docker.sh + ./rvs_nightly_test.sh setup-ssh + + - name: Run RVS level 4 in docker on target + id: level4 + run: ./rvs_nightly_docker.sh run-level4 + + - name: Capture RVS and ROCm versions from docker on target + id: versions + if: always() + run: ./rvs_nightly_docker.sh capture-versions + + - name: Collect logs from target + if: always() + run: ./rvs_nightly_test.sh collect-logs + + - name: Upload test logs + if: always() + uses: actions/upload-artifact@v4 + with: + name: rvs-nightly-docker-rhel9-logs-${{ github.run_id }} + path: ./reports/ + retention-days: 30 + if-no-files-found: warn + + create-test-report: + name: Create Test Report + needs: [install-rvs-in-docker, run-rvs-level-4-in-docker] + runs-on: ${{ vars.RUNNER_LABEL_UTILITY || 'ubuntu-latest' }} + if: always() && !cancelled() && needs.install-rvs-in-docker.result == 'success' && needs.run-rvs-level-4-in-docker.result != 'skipped' + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + + - name: Download test logs + uses: actions/download-artifact@v4 + with: + name: rvs-nightly-docker-rhel9-logs-${{ github.run_id }} + path: ./reports + continue-on-error: true + + - name: Build test report + id: report + env: + TARBALL_URL: ${{ needs.install-rvs-in-docker.outputs.tarball_url }} + TARBALL_NAME: ${{ needs.install-rvs-in-docker.outputs.tarball_name }} + TARGET_ROCM_PATH: /opt/rocm + REMOTE_WORK_DIR: ${{ needs.install-rvs-in-docker.outputs.remote_work_dir }} + RVS_BIN: ${{ needs.install-rvs-in-docker.outputs.rvs_bin }} + RVS_LEVEL4_RC: ${{ needs.run-rvs-level-4-in-docker.outputs.rc }} + RVS_LEVEL4_START: ${{ needs.run-rvs-level-4-in-docker.outputs.start }} + RVS_LEVEL4_END: ${{ needs.run-rvs-level-4-in-docker.outputs.end }} + RVS_VERSION: ${{ needs.run-rvs-level-4-in-docker.outputs.rvs_version }} + TARGET_ROCM_VERSION: ${{ needs.run-rvs-level-4-in-docker.outputs.target_rocm_version }} + GITHUB_RUN_ID: ${{ github.run_id }} + GITHUB_SERVER_URL: ${{ github.server_url }} + GITHUB_REPOSITORY: ${{ github.repository }} + GITHUB_EVENT_NAME: ${{ inputs.trigger_event || github.event_name }} + run: | + chmod +x rvs_nightly_test.sh + mkdir -p ./reports + ./rvs_nightly_test.sh build-report + + - name: Upload report and logs + if: always() + uses: actions/upload-artifact@v4 + with: + name: rvs-nightly-docker-rhel9-report-${{ github.run_id }} + path: ./reports/ + retention-days: 30 + if-no-files-found: warn + + - name: Fail the workflow if level 4 failed + if: steps.report.outputs.overall == 'FAIL' || needs.run-rvs-level-4-in-docker.result == 'failure' + run: | + echo "::error::RVS docker (RHEL 9) test failure — level4 rc=${{ needs.run-rvs-level-4-in-docker.outputs.rc }}" + exit 1 diff --git a/.github/workflows/rvs-nightly-docker-ubuntu22.04-tests.yml b/.github/workflows/rvs-nightly-docker-ubuntu22.04-tests.yml new file mode 100644 index 000000000..3ef184bc8 --- /dev/null +++ b/.github/workflows/rvs-nightly-docker-ubuntu22.04-tests.yml @@ -0,0 +1,346 @@ +name: RVS Nightly Tests (Docker Ubuntu 22.04) + +# Same flow as rvs-nightly-docker-ubuntu24.04-tests.yml but the ROCm runtime container uses ubuntu:22.04. +# ROCm SDK version is derived from the tar tarball name (-rMMmm.yyyymmdd-Linux.tar.gz). +# +# Default image delivery: build-on-target (docker build on the GPU node; no cross-host image transfer). +# Set workflow input build_on_target=false to fall back to scp/registry delivery from the runner. +# +# Image delivery (RVS_DOCKER_TRANSFER_MODE / workflow input docker_transfer_mode): +# auto (default) — build-on-target when build_on_target is true (default), else registry when +# RVS_DOCKER_REGISTRY is set, else scp (docker save | pigz/gzip | scp | docker load) +# scp — always save/scp/load (skip when target already has matching ROCM_VERSION) +# registry — docker push on build host, docker pull on target (requires RVS_DOCKER_REGISTRY) +# build-on-target — docker build on the GPU node (no cross-host image transfer) +# +# RVS tarball: downloaded on the orchestrator runner and scp'd to the target (not wget in container). +# +# Prerequisites: +# - Build host (self-hosted runner): curl/scp/ssh; HTTPS egress to tarball index; checkout supplies docker build context +# - Target GPU node: docker, HTTPS to nightly.repo.amd.com, /dev/kfd, /dev/dri, SSH user in docker group +# - ROCm SDK in docker image: therock-dist-linux-multiarch-.tar.gz on ubuntu:22.04 +# - Secrets: RVS_TARGET_NODE, RVS_TARGET_SSH_KEY (optional RVS_TARGET_USER, RVS_TARBALL_INDEX_URL) +# - Optional: vars.ROCM_SDK_NIGHTLY_BASE_URL to override the ROCm SDK nightly tarball host +# - Optional: vars.RVS_DOCKER_REGISTRY + secrets RVS_DOCKER_REGISTRY_USER/PASSWORD for registry mode +# +# See README_NIGHTLY_TESTS.md for SSH / tarball index configuration. + +on: + workflow_dispatch: + inputs: + trigger_event: + description: 'Label for report only (e.g. schedule or push); optional' + required: false + default: '' + tarball_url: + description: 'Override tarball URL (default: latest from index — secrets.RVS_TARBALL_INDEX_URL)' + required: false + default: '' + target_node: + description: 'GPU target hostname/IP (default: secrets.RVS_TARGET_NODE)' + required: false + default: '' + target_user: + description: 'SSH user on target (default: secrets.RVS_TARGET_USER)' + required: false + default: '' + docker_image: + description: 'ROCm runtime image (default: vars.RVS_NIGHTLY_DOCKER_UBUNTU22_IMAGE or rvs-nightly-rocm-ubuntu22.04:latest)' + required: false + default: '' + build_docker_image: + description: 'Build rvs-nightly-rocm-ubuntu22.04 on the build host before scp/registry delivery (only when build_on_target is false)' + required: false + type: boolean + default: false + build_on_target: + description: 'Build the ROCm docker image on the GPU target (default; skips cross-host image transfer)' + required: false + type: boolean + default: true + docker_transfer_mode: + description: 'Image delivery — auto, scp, registry, or build-on-target (default auto)' + required: false + default: 'auto' + remote_work_dir: + description: 'Work dir on target (default: /tmp/rvs-nightly-docker-ubuntu22.04-)' + required: false + default: '' + fallback_latest_sdk: + description: 'If the exact SDK date is missing, fall back to latest same line, then same major, then newest nightly' + required: false + type: boolean + default: true + +permissions: + contents: read + +concurrency: + group: rvs-nightly-docker-ubuntu22.04-${{ github.workflow }} + cancel-in-progress: false + +env: + TARBALL_INDEX_URL: ${{ secrets.RVS_TARBALL_INDEX_URL }} + RVS_NIGHTLY_DOCKER_IMAGE: ${{ inputs.docker_image || vars.RVS_NIGHTLY_DOCKER_UBUNTU22_IMAGE || 'rvs-nightly-rocm-ubuntu22.04:latest' }} + RVS_DOCKER_BUILD_DIR: .github/docker/rvs-nightly-rocm-ubuntu22.04 + RVS_DOCKER_IMAGE_ARCHIVE: rvs-nightly-rocm-ubuntu22.04-image.tar.gz + RVS_DOCKER_ON_TARGET: 'true' + RVS_DOCKER_TRANSFER_MODE: ${{ inputs.docker_transfer_mode || vars.RVS_DOCKER_TRANSFER_MODE || 'auto' }} + RVS_DOCKER_BUILD_ON_TARGET: ${{ (inputs.build_on_target != false) && (vars.RVS_DOCKER_BUILD_ON_TARGET != 'false') && 'true' || 'false' }} + RVS_DOCKER_REGISTRY: ${{ vars.RVS_DOCKER_REGISTRY || '' }} + RVS_DOCKER_REGISTRY_USER: ${{ secrets.RVS_DOCKER_REGISTRY_USER }} + RVS_DOCKER_REGISTRY_PASSWORD: ${{ secrets.RVS_DOCKER_REGISTRY_PASSWORD }} + RVS_DOCKER_SKIP_IF_PRESENT: 'true' + # Default true on schedule (inputs unset). github.event.inputs is a string; empty != 'false'. + # Opt out: uncheck workflow_dispatch fallback_latest_sdk, or set vars.RVS_DOCKER_SDK_FALLBACK_LATEST=false + RVS_DOCKER_SDK_FALLBACK_LATEST: ${{ (github.event.inputs.fallback_latest_sdk != 'false' && vars.RVS_DOCKER_SDK_FALLBACK_LATEST != 'false') && 'true' || 'false' }} + # ROCm SDK nightly host. Set vars.ROCM_SDK_NIGHTLY_BASE_URL (e.g. https://nightly.repo.amd.com/rocm/core/tarball). + # Optional vars.ROCM_SDK_NIGHTLY_INDEX_URL; if unset, BASE_URL is used for both listing and download. + ROCM_SDK_NIGHTLY_BASE_URL: ${{ vars.ROCM_SDK_NIGHTLY_BASE_URL || '' }} + ROCM_SDK_NIGHTLY_INDEX_URL: ${{ vars.ROCM_SDK_NIGHTLY_INDEX_URL || vars.ROCM_SDK_NIGHTLY_BASE_URL || '' }} + TARGET_ROCM_PATH: /opt/rocm/install + TARGET_NODE: ${{ inputs.target_node || secrets.RVS_TARGET_NODE }} + TARGET_USER: ${{ inputs.target_user || secrets.RVS_TARGET_USER }} + SSH_PRIVATE_KEY: ${{ secrets.RVS_TARGET_SSH_KEY }} + +jobs: + install-rvs-in-docker: + name: Ensure ROCm image (Ubuntu 22.04) on target and install RVS in docker + runs-on: ${{ vars.RVS_NIGHTLY_DOCKER_RUNNER_LABEL || vars.RVS_TEST_RUNNER_LABEL || 'self-hosted' }} + timeout-minutes: 180 + outputs: + tarball_url: ${{ steps.resolve.outputs.tarball_url }} + tarball_name: ${{ steps.resolve.outputs.tarball_name }} + remote_work_dir: ${{ steps.prepare.outputs.remote_work_dir }} + rocm_major: ${{ steps.prepare.outputs.rocm_major }} + install_dir: ${{ steps.prepare.outputs.install_dir }} + rvs_bin: ${{ steps.prepare.outputs.rvs_bin }} + env: + TARGET_ROCM_PATH: /opt/rocm/install + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + + - name: Resolve latest tarball URL + id: resolve + env: + INPUT_URL: ${{ inputs.tarball_url }} + run: | + set -euo pipefail + INDEX_URL="${TARBALL_INDEX_URL:-}" + if [ -z "${INPUT_URL:-}" ] && [ -z "${INDEX_URL}" ]; then + echo "::error::No tarball index URL (set RVS_TARBALL_INDEX_URL) and no tarball_url input." + exit 1 + fi + fetch_latest() { + local idx="$1" + local html raw name + html="$(curl -sSL --max-time 60 --retry 2 --retry-delay 2 "$idx" 2>/dev/null || true)" + raw="$(printf '%s' "$html" | grep -oE 'amdrocm[0-9]*-rvs-[0-9A-Za-z._\-]+-Linux\.tar\.gz' 2>/dev/null || true)" + name="$(printf '%s\n' "$raw" | sort -uV | tail -n 1 || true)" + [ -n "$name" ] || return 1 + printf '%s/%s\n' "${idx%/}" "$name" + } + if [ -n "${INPUT_URL:-}" ]; then + URL="$INPUT_URL" + else + URL="$(fetch_latest "$INDEX_URL")" + fi + [ -n "${URL:-}" ] || { echo "::error::Could not resolve tarball URL"; exit 1; } + NAME="$(basename "$URL" | tr -d '\r\n')" + if [[ "$NAME" != *-Linux.tar.gz ]]; then + echo "::error::Docker nightly tests require a *-Linux.tar.gz relocatable tarball; got: ${NAME}" >&2 + exit 1 + fi + if [[ ! "$NAME" =~ -r[0-9]{4}\.[0-9]{8}-Linux\.tar\.gz$ ]]; then + echo "::error::Tar tarball must include -rMMmm.yyyymmdd- (e.g. -r0715.20260724-Linux.tar.gz); got: ${NAME}" >&2 + exit 1 + fi + { + echo "tarball_url<> "$GITHUB_OUTPUT" + { + echo "TARBALL_URL<> "$GITHUB_ENV" + echo "Latest tarball : $NAME" + + - name: Validate configuration and derive paths + id: prepare + env: + TARBALL_NAME: ${{ steps.resolve.outputs.tarball_name }} + INPUT_REMOTE_WORK_DIR: ${{ inputs.remote_work_dir || format('/tmp/rvs-nightly-docker-ubuntu22.04-{0}', github.run_id) }} + VAR_REMOTE_WORK_DIR: ${{ vars.RVS_REMOTE_WORK_DIR }} + GITHUB_RUN_ID: ${{ github.run_id }} + run: | + chmod +x rvs_nightly_test.sh + if [ -z "${TARGET_NODE:-}" ]; then + echo "::error::No target node (set secrets.RVS_TARGET_NODE or workflow input target_node)." >&2 + exit 1 + fi + ./rvs_nightly_test.sh validate-config + + - name: Setup SSH to target node + env: + REMOTE_WORK_DIR: ${{ steps.prepare.outputs.remote_work_dir }} + run: ./rvs_nightly_test.sh setup-ssh + + - name: Export paths for remaining steps + env: + TARBALL_NAME: ${{ steps.resolve.outputs.tarball_name }} + run: | + { + echo "REMOTE_WORK_DIR=${{ steps.prepare.outputs.remote_work_dir }}" + echo "ROCM_MAJOR=${{ steps.prepare.outputs.rocm_major }}" + echo "INSTALL_DIR=${{ steps.prepare.outputs.install_dir }}" + echo "RVS_BIN=${{ steps.prepare.outputs.rvs_bin }}" + } >> "$GITHUB_ENV" + if [[ "$TARBALL_NAME" =~ -r([0-9]{2})([0-9]{2})\.([0-9]{8})-Linux\.tar\.gz$ ]]; then + echo "RVS_DOCKER_ROCM_VERSION=$((10#${BASH_REMATCH[1]})).$((10#${BASH_REMATCH[2]})).0a${BASH_REMATCH[3]}" >> "$GITHUB_ENV" + fi + + - name: Build ROCm docker image on build host + if: inputs.build_docker_image && inputs.build_on_target == false + env: + TARBALL_NAME: ${{ steps.resolve.outputs.tarball_name }} + run: | + chmod +x .github/docker/build-rocm-sdk-image.sh + fallback_args=() + if [ "${RVS_DOCKER_SDK_FALLBACK_LATEST}" = true ]; then + fallback_args=(--fallback-latest-sdk) + fi + ./.github/docker/build-rocm-sdk-image.sh --context .github/docker/rvs-nightly-rocm-ubuntu22.04 --from-tarball "$TARBALL_NAME" "${fallback_args[@]}" + + - name: Verify ROCm docker image on build host + if: inputs.build_docker_image && inputs.build_on_target == false + run: | + chmod +x rvs_nightly_docker.sh + ./rvs_nightly_docker.sh pull-image + + - name: Ensure ROCm docker image on GPU target + env: + TARBALL_NAME: ${{ steps.resolve.outputs.tarball_name }} + run: | + chmod +x rvs_nightly_docker.sh + ./rvs_nightly_docker.sh ensure-image-on-target + + - name: Verify ROCm in docker on target + run: ./rvs_nightly_docker.sh verify-rocm + + - name: Install RVS in docker on target + run: ./rvs_nightly_docker.sh install-rvs + + run-rvs-level-4-in-docker: + name: Run RVS level 4 in docker (Ubuntu 22.04) on target + needs: [install-rvs-in-docker] + runs-on: ${{ vars.RVS_NIGHTLY_DOCKER_RUNNER_LABEL || vars.RVS_TEST_RUNNER_LABEL || 'self-hosted' }} + timeout-minutes: 180 + outputs: + rc: ${{ steps.level4.outputs.rc }} + start: ${{ steps.level4.outputs.start }} + end: ${{ steps.level4.outputs.end }} + rvs_version: ${{ steps.versions.outputs.rvs_version }} + target_rocm_version: ${{ steps.versions.outputs.target_rocm_version }} + env: + TARGET_ROCM_PATH: /opt/rocm/install + TARGET_NODE: ${{ inputs.target_node || secrets.RVS_TARGET_NODE }} + TARGET_USER: ${{ inputs.target_user || secrets.RVS_TARGET_USER }} + SSH_PRIVATE_KEY: ${{ secrets.RVS_TARGET_SSH_KEY }} + RVS_DOCKER_ON_TARGET: 'true' + RVS_DOCKER_BUILD_DIR: .github/docker/rvs-nightly-rocm-ubuntu22.04 + RVS_DOCKER_IMAGE_ARCHIVE: rvs-nightly-rocm-ubuntu22.04-image.tar.gz + REMOTE_WORK_DIR: ${{ needs.install-rvs-in-docker.outputs.remote_work_dir }} + ROCM_MAJOR: ${{ needs.install-rvs-in-docker.outputs.rocm_major }} + INSTALL_DIR: ${{ needs.install-rvs-in-docker.outputs.install_dir }} + RVS_BIN: ${{ needs.install-rvs-in-docker.outputs.rvs_bin }} + RVS_NIGHTLY_DOCKER_IMAGE: ${{ inputs.docker_image || vars.RVS_NIGHTLY_DOCKER_UBUNTU22_IMAGE || 'rvs-nightly-rocm-ubuntu22.04:latest' }} + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + + - name: Setup SSH to target node + env: + REMOTE_WORK_DIR: ${{ needs.install-rvs-in-docker.outputs.remote_work_dir }} + run: | + chmod +x rvs_nightly_test.sh rvs_nightly_docker.sh + ./rvs_nightly_test.sh setup-ssh + + - name: Run RVS level 4 in docker on target + id: level4 + run: ./rvs_nightly_docker.sh run-level4 + + - name: Capture RVS and ROCm versions from docker on target + id: versions + if: always() + run: ./rvs_nightly_docker.sh capture-versions + + - name: Collect logs from target + if: always() + run: ./rvs_nightly_test.sh collect-logs + + - name: Upload test logs + if: always() + uses: actions/upload-artifact@v4 + with: + name: rvs-nightly-docker-ubuntu22.04-logs-${{ github.run_id }} + path: ./reports/ + retention-days: 30 + if-no-files-found: warn + + create-test-report: + name: Create Test Report + needs: [install-rvs-in-docker, run-rvs-level-4-in-docker] + runs-on: ${{ vars.RUNNER_LABEL_UTILITY || 'ubuntu-latest' }} + if: always() && !cancelled() && needs.install-rvs-in-docker.result == 'success' && needs.run-rvs-level-4-in-docker.result != 'skipped' + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + + - name: Download test logs + uses: actions/download-artifact@v4 + with: + name: rvs-nightly-docker-ubuntu22.04-logs-${{ github.run_id }} + path: ./reports + continue-on-error: true + + - name: Build test report + id: report + env: + TARBALL_URL: ${{ needs.install-rvs-in-docker.outputs.tarball_url }} + TARBALL_NAME: ${{ needs.install-rvs-in-docker.outputs.tarball_name }} + TARGET_ROCM_PATH: /opt/rocm/install + REMOTE_WORK_DIR: ${{ needs.install-rvs-in-docker.outputs.remote_work_dir }} + RVS_BIN: ${{ needs.install-rvs-in-docker.outputs.rvs_bin }} + RVS_LEVEL4_RC: ${{ needs.run-rvs-level-4-in-docker.outputs.rc }} + RVS_LEVEL4_START: ${{ needs.run-rvs-level-4-in-docker.outputs.start }} + RVS_LEVEL4_END: ${{ needs.run-rvs-level-4-in-docker.outputs.end }} + RVS_VERSION: ${{ needs.run-rvs-level-4-in-docker.outputs.rvs_version }} + TARGET_ROCM_VERSION: ${{ needs.run-rvs-level-4-in-docker.outputs.target_rocm_version }} + GITHUB_RUN_ID: ${{ github.run_id }} + GITHUB_SERVER_URL: ${{ github.server_url }} + GITHUB_REPOSITORY: ${{ github.repository }} + GITHUB_EVENT_NAME: ${{ inputs.trigger_event || github.event_name }} + run: | + chmod +x rvs_nightly_test.sh + mkdir -p ./reports + ./rvs_nightly_test.sh build-report + + - name: Upload report and logs + if: always() + uses: actions/upload-artifact@v4 + with: + name: rvs-nightly-docker-ubuntu22.04-report-${{ github.run_id }} + path: ./reports/ + retention-days: 30 + if-no-files-found: warn + + - name: Fail the workflow if level 4 failed + if: steps.report.outputs.overall == 'FAIL' || needs.run-rvs-level-4-in-docker.result == 'failure' + run: | + echo "::error::RVS docker (Ubuntu 22.04) test failure — level4 rc=${{ needs.run-rvs-level-4-in-docker.outputs.rc }}" + exit 1 diff --git a/.github/workflows/rvs-nightly-docker-ubuntu24.04-tests.yml b/.github/workflows/rvs-nightly-docker-ubuntu24.04-tests.yml new file mode 100644 index 000000000..630df21e5 --- /dev/null +++ b/.github/workflows/rvs-nightly-docker-ubuntu24.04-tests.yml @@ -0,0 +1,356 @@ +name: RVS Nightly Tests (Docker Ubuntu 24.04) + +# Deliver a ROCm-matched docker image (ubuntu:24.04) to the GPU target, install RVS, run level 4 in docker. +# ROCm SDK version is derived from the tar tarball name (-rMMmm.yyyymmdd-Linux.tar.gz). +# +# Default image delivery: build-on-target (docker build on the GPU node; no cross-host image transfer). +# Set workflow input build_on_target=false to fall back to scp/registry delivery from the runner. +# +# Image delivery (RVS_DOCKER_TRANSFER_MODE / workflow input docker_transfer_mode): +# auto (default) — build-on-target when build_on_target is true (default), else registry when +# RVS_DOCKER_REGISTRY is set, else scp (docker save | pigz/gzip | scp | docker load) +# scp — always save/scp/load (skip when target already has matching ROCM_VERSION) +# registry — docker push on build host, docker pull on target (requires RVS_DOCKER_REGISTRY) +# build-on-target — docker build on the GPU node (no cross-host image transfer) +# +# RVS tarball: downloaded on the orchestrator runner and scp'd to the target (not wget in container). +# +# Prerequisites: +# - Build host (self-hosted runner): curl/scp/ssh; HTTPS egress to tarball index; checkout supplies docker build context +# - Target GPU node: docker, HTTPS to nightly.repo.amd.com, /dev/kfd, /dev/dri, SSH user in docker group +# - ROCm SDK in docker image: therock-dist-linux-multiarch-.tar.gz on ubuntu:24.04 +# - Secrets: RVS_TARGET_NODE, RVS_TARGET_SSH_KEY (optional RVS_TARGET_USER, RVS_TARBALL_INDEX_URL) +# - Optional: vars.ROCM_SDK_NIGHTLY_BASE_URL to override the ROCm SDK nightly tarball host +# - Optional: vars.RVS_DOCKER_REGISTRY + secrets RVS_DOCKER_REGISTRY_USER/PASSWORD for registry mode +# +# See README_NIGHTLY_TESTS.md for SSH / tarball index configuration. +# +# Triggers +# -------- +# - schedule: daily at 8:00 AM PST (16:00 UTC; PST is UTC-8) +# - workflow_dispatch: manual run, or dispatched by Build Relocatable Packages +# after a successful push to main/master +# - trigger_event input (push) is set by the package-build dispatch step + +on: + schedule: + # Run daily at 8:00 AM PST (16:00 UTC; PST is UTC-8) + - cron: '0 16 * * *' + workflow_dispatch: + inputs: + trigger_event: + description: 'Package-build event that triggered this run (push); empty for manual or schedule runs' + required: false + default: '' + tarball_url: + description: 'Override tarball URL (default: latest from index — secrets.RVS_TARBALL_INDEX_URL)' + required: false + default: '' + target_node: + description: 'GPU target hostname/IP (default: secrets.RVS_TARGET_NODE)' + required: false + default: '' + target_user: + description: 'SSH user on target (default: secrets.RVS_TARGET_USER)' + required: false + default: '' + docker_image: + description: 'ROCm runtime image (default: vars.RVS_NIGHTLY_DOCKER_UBUNTU24_IMAGE or vars.RVS_NIGHTLY_DOCKER_IMAGE or rvs-nightly-rocm:latest)' + required: false + default: '' + build_docker_image: + description: 'Build rvs-nightly-rocm on the build host before scp/registry delivery (only when build_on_target is false)' + required: false + type: boolean + default: false + build_on_target: + description: 'Build the ROCm docker image on the GPU target (default; skips cross-host image transfer)' + required: false + type: boolean + default: true + docker_transfer_mode: + description: 'Image delivery — auto, scp, registry, or build-on-target (default auto)' + required: false + default: 'auto' + remote_work_dir: + description: 'Work dir on target (default: /tmp/rvs-nightly-docker-ubuntu24.04-)' + required: false + default: '' + fallback_latest_sdk: + description: 'If the exact SDK date is missing, fall back to latest same line, then same major, then newest nightly' + required: false + type: boolean + default: true + +permissions: + contents: read + +concurrency: + group: rvs-nightly-docker-ubuntu24.04-${{ github.workflow }} + cancel-in-progress: false + +env: + TARBALL_INDEX_URL: ${{ secrets.RVS_TARBALL_INDEX_URL }} + RVS_NIGHTLY_DOCKER_IMAGE: ${{ inputs.docker_image || vars.RVS_NIGHTLY_DOCKER_UBUNTU24_IMAGE || vars.RVS_NIGHTLY_DOCKER_IMAGE || 'rvs-nightly-rocm:latest' }} + RVS_DOCKER_BUILD_DIR: .github/docker/rvs-nightly-rocm + RVS_DOCKER_IMAGE_ARCHIVE: rvs-nightly-rocm-ubuntu24.04-image.tar.gz + RVS_DOCKER_ON_TARGET: 'true' + RVS_DOCKER_TRANSFER_MODE: ${{ inputs.docker_transfer_mode || vars.RVS_DOCKER_TRANSFER_MODE || 'auto' }} + RVS_DOCKER_BUILD_ON_TARGET: ${{ (inputs.build_on_target != false) && (vars.RVS_DOCKER_BUILD_ON_TARGET != 'false') && 'true' || 'false' }} + RVS_DOCKER_REGISTRY: ${{ vars.RVS_DOCKER_REGISTRY || '' }} + RVS_DOCKER_REGISTRY_USER: ${{ secrets.RVS_DOCKER_REGISTRY_USER }} + RVS_DOCKER_REGISTRY_PASSWORD: ${{ secrets.RVS_DOCKER_REGISTRY_PASSWORD }} + RVS_DOCKER_SKIP_IF_PRESENT: 'true' + # Default true on schedule (inputs unset). github.event.inputs is a string; empty != 'false'. + # Opt out: uncheck workflow_dispatch fallback_latest_sdk, or set vars.RVS_DOCKER_SDK_FALLBACK_LATEST=false + RVS_DOCKER_SDK_FALLBACK_LATEST: ${{ (github.event.inputs.fallback_latest_sdk != 'false' && vars.RVS_DOCKER_SDK_FALLBACK_LATEST != 'false') && 'true' || 'false' }} + # ROCm SDK nightly host. Set vars.ROCM_SDK_NIGHTLY_BASE_URL (e.g. https://nightly.repo.amd.com/rocm/core/tarball). + # Optional vars.ROCM_SDK_NIGHTLY_INDEX_URL; if unset, BASE_URL is used for both listing and download. + ROCM_SDK_NIGHTLY_BASE_URL: ${{ vars.ROCM_SDK_NIGHTLY_BASE_URL || '' }} + ROCM_SDK_NIGHTLY_INDEX_URL: ${{ vars.ROCM_SDK_NIGHTLY_INDEX_URL || vars.ROCM_SDK_NIGHTLY_BASE_URL || '' }} + TARGET_ROCM_PATH: /opt/rocm/install + TARGET_NODE: ${{ inputs.target_node || secrets.RVS_TARGET_NODE }} + TARGET_USER: ${{ inputs.target_user || secrets.RVS_TARGET_USER }} + SSH_PRIVATE_KEY: ${{ secrets.RVS_TARGET_SSH_KEY }} + +jobs: + install-rvs-in-docker: + name: Ensure ROCm image (Ubuntu 24.04) on target and install RVS in docker + runs-on: ${{ vars.RVS_NIGHTLY_DOCKER_RUNNER_LABEL || vars.RVS_TEST_RUNNER_LABEL || 'self-hosted' }} + timeout-minutes: 180 + outputs: + tarball_url: ${{ steps.resolve.outputs.tarball_url }} + tarball_name: ${{ steps.resolve.outputs.tarball_name }} + remote_work_dir: ${{ steps.prepare.outputs.remote_work_dir }} + rocm_major: ${{ steps.prepare.outputs.rocm_major }} + install_dir: ${{ steps.prepare.outputs.install_dir }} + rvs_bin: ${{ steps.prepare.outputs.rvs_bin }} + env: + TARGET_ROCM_PATH: /opt/rocm/install + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + + - name: Resolve latest tarball URL + id: resolve + env: + INPUT_URL: ${{ inputs.tarball_url }} + run: | + set -euo pipefail + INDEX_URL="${TARBALL_INDEX_URL:-}" + if [ -z "${INPUT_URL:-}" ] && [ -z "${INDEX_URL}" ]; then + echo "::error::No tarball index URL (set RVS_TARBALL_INDEX_URL) and no tarball_url input." + exit 1 + fi + fetch_latest() { + local idx="$1" + local html raw name + html="$(curl -sSL --max-time 60 --retry 2 --retry-delay 2 "$idx" 2>/dev/null || true)" + raw="$(printf '%s' "$html" | grep -oE 'amdrocm[0-9]*-rvs-[0-9A-Za-z._\-]+-Linux\.tar\.gz' 2>/dev/null || true)" + name="$(printf '%s\n' "$raw" | sort -uV | tail -n 1 || true)" + [ -n "$name" ] || return 1 + printf '%s/%s\n' "${idx%/}" "$name" + } + if [ -n "${INPUT_URL:-}" ]; then + URL="$INPUT_URL" + else + URL="$(fetch_latest "$INDEX_URL")" + fi + [ -n "${URL:-}" ] || { echo "::error::Could not resolve tarball URL"; exit 1; } + NAME="$(basename "$URL" | tr -d '\r\n')" + if [[ "$NAME" != *-Linux.tar.gz ]]; then + echo "::error::Docker nightly tests require a *-Linux.tar.gz relocatable tarball; got: ${NAME}" >&2 + exit 1 + fi + if [[ ! "$NAME" =~ -r[0-9]{4}\.[0-9]{8}-Linux\.tar\.gz$ ]]; then + echo "::error::Tar tarball must include -rMMmm.yyyymmdd- (e.g. -r0715.20260724-Linux.tar.gz); got: ${NAME}" >&2 + exit 1 + fi + { + echo "tarball_url<> "$GITHUB_OUTPUT" + { + echo "TARBALL_URL<> "$GITHUB_ENV" + echo "Latest tarball : $NAME" + + - name: Validate configuration and derive paths + id: prepare + env: + TARBALL_NAME: ${{ steps.resolve.outputs.tarball_name }} + INPUT_REMOTE_WORK_DIR: ${{ inputs.remote_work_dir || format('/tmp/rvs-nightly-docker-ubuntu24.04-{0}', github.run_id) }} + VAR_REMOTE_WORK_DIR: ${{ vars.RVS_REMOTE_WORK_DIR }} + GITHUB_RUN_ID: ${{ github.run_id }} + run: | + chmod +x rvs_nightly_test.sh + if [ -z "${TARGET_NODE:-}" ]; then + echo "::error::No target node (set secrets.RVS_TARGET_NODE or workflow input target_node)." >&2 + exit 1 + fi + ./rvs_nightly_test.sh validate-config + + - name: Setup SSH to target node + env: + REMOTE_WORK_DIR: ${{ steps.prepare.outputs.remote_work_dir }} + run: ./rvs_nightly_test.sh setup-ssh + + - name: Export paths for remaining steps + env: + TARBALL_NAME: ${{ steps.resolve.outputs.tarball_name }} + run: | + { + echo "REMOTE_WORK_DIR=${{ steps.prepare.outputs.remote_work_dir }}" + echo "ROCM_MAJOR=${{ steps.prepare.outputs.rocm_major }}" + echo "INSTALL_DIR=${{ steps.prepare.outputs.install_dir }}" + echo "RVS_BIN=${{ steps.prepare.outputs.rvs_bin }}" + } >> "$GITHUB_ENV" + if [[ "$TARBALL_NAME" =~ -r([0-9]{2})([0-9]{2})\.([0-9]{8})-Linux\.tar\.gz$ ]]; then + echo "RVS_DOCKER_ROCM_VERSION=$((10#${BASH_REMATCH[1]})).$((10#${BASH_REMATCH[2]})).0a${BASH_REMATCH[3]}" >> "$GITHUB_ENV" + fi + + - name: Build ROCm docker image on build host + if: inputs.build_docker_image && inputs.build_on_target == false + env: + TARBALL_NAME: ${{ steps.resolve.outputs.tarball_name }} + run: | + chmod +x .github/docker/build-rocm-sdk-image.sh + fallback_args=() + if [ "${RVS_DOCKER_SDK_FALLBACK_LATEST}" = true ]; then + fallback_args=(--fallback-latest-sdk) + fi + ./.github/docker/build-rocm-sdk-image.sh --context .github/docker/rvs-nightly-rocm --from-tarball "$TARBALL_NAME" "${fallback_args[@]}" + + - name: Verify ROCm docker image on build host + if: inputs.build_docker_image && inputs.build_on_target == false + run: | + chmod +x rvs_nightly_docker.sh + ./rvs_nightly_docker.sh pull-image + + - name: Ensure ROCm docker image on GPU target + env: + TARBALL_NAME: ${{ steps.resolve.outputs.tarball_name }} + run: | + chmod +x rvs_nightly_docker.sh + ./rvs_nightly_docker.sh ensure-image-on-target + + - name: Verify ROCm in docker on target + run: ./rvs_nightly_docker.sh verify-rocm + + - name: Install RVS in docker on target + run: ./rvs_nightly_docker.sh install-rvs + + run-rvs-level-4-in-docker: + name: Run RVS level 4 in docker (Ubuntu 24.04) on target + needs: [install-rvs-in-docker] + runs-on: ${{ vars.RVS_NIGHTLY_DOCKER_RUNNER_LABEL || vars.RVS_TEST_RUNNER_LABEL || 'self-hosted' }} + timeout-minutes: 180 + outputs: + rc: ${{ steps.level4.outputs.rc }} + start: ${{ steps.level4.outputs.start }} + end: ${{ steps.level4.outputs.end }} + rvs_version: ${{ steps.versions.outputs.rvs_version }} + target_rocm_version: ${{ steps.versions.outputs.target_rocm_version }} + env: + TARGET_ROCM_PATH: /opt/rocm/install + TARGET_NODE: ${{ inputs.target_node || secrets.RVS_TARGET_NODE }} + TARGET_USER: ${{ inputs.target_user || secrets.RVS_TARGET_USER }} + SSH_PRIVATE_KEY: ${{ secrets.RVS_TARGET_SSH_KEY }} + RVS_DOCKER_ON_TARGET: 'true' + RVS_DOCKER_BUILD_DIR: .github/docker/rvs-nightly-rocm + RVS_DOCKER_IMAGE_ARCHIVE: rvs-nightly-rocm-ubuntu24.04-image.tar.gz + REMOTE_WORK_DIR: ${{ needs.install-rvs-in-docker.outputs.remote_work_dir }} + ROCM_MAJOR: ${{ needs.install-rvs-in-docker.outputs.rocm_major }} + INSTALL_DIR: ${{ needs.install-rvs-in-docker.outputs.install_dir }} + RVS_BIN: ${{ needs.install-rvs-in-docker.outputs.rvs_bin }} + RVS_NIGHTLY_DOCKER_IMAGE: ${{ inputs.docker_image || vars.RVS_NIGHTLY_DOCKER_UBUNTU24_IMAGE || vars.RVS_NIGHTLY_DOCKER_IMAGE || 'rvs-nightly-rocm:latest' }} + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + + - name: Setup SSH to target node + env: + REMOTE_WORK_DIR: ${{ needs.install-rvs-in-docker.outputs.remote_work_dir }} + run: | + chmod +x rvs_nightly_test.sh rvs_nightly_docker.sh + ./rvs_nightly_test.sh setup-ssh + + - name: Run RVS level 4 in docker on target + id: level4 + run: ./rvs_nightly_docker.sh run-level4 + + - name: Capture RVS and ROCm versions from docker on target + id: versions + if: always() + run: ./rvs_nightly_docker.sh capture-versions + + - name: Collect logs from target + if: always() + run: ./rvs_nightly_test.sh collect-logs + + - name: Upload test logs + if: always() + uses: actions/upload-artifact@v4 + with: + name: rvs-nightly-docker-ubuntu24.04-logs-${{ github.run_id }} + path: ./reports/ + retention-days: 30 + if-no-files-found: warn + + create-test-report: + name: Create Test Report + needs: [install-rvs-in-docker, run-rvs-level-4-in-docker] + runs-on: ${{ vars.RUNNER_LABEL_UTILITY || 'ubuntu-latest' }} + if: always() && !cancelled() && needs.install-rvs-in-docker.result == 'success' && needs.run-rvs-level-4-in-docker.result != 'skipped' + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + + - name: Download test logs + uses: actions/download-artifact@v4 + with: + name: rvs-nightly-docker-ubuntu24.04-logs-${{ github.run_id }} + path: ./reports + continue-on-error: true + + - name: Build test report + id: report + env: + TARBALL_URL: ${{ needs.install-rvs-in-docker.outputs.tarball_url }} + TARBALL_NAME: ${{ needs.install-rvs-in-docker.outputs.tarball_name }} + TARGET_ROCM_PATH: /opt/rocm/install + REMOTE_WORK_DIR: ${{ needs.install-rvs-in-docker.outputs.remote_work_dir }} + RVS_BIN: ${{ needs.install-rvs-in-docker.outputs.rvs_bin }} + RVS_LEVEL4_RC: ${{ needs.run-rvs-level-4-in-docker.outputs.rc }} + RVS_LEVEL4_START: ${{ needs.run-rvs-level-4-in-docker.outputs.start }} + RVS_LEVEL4_END: ${{ needs.run-rvs-level-4-in-docker.outputs.end }} + RVS_VERSION: ${{ needs.run-rvs-level-4-in-docker.outputs.rvs_version }} + TARGET_ROCM_VERSION: ${{ needs.run-rvs-level-4-in-docker.outputs.target_rocm_version }} + GITHUB_RUN_ID: ${{ github.run_id }} + GITHUB_SERVER_URL: ${{ github.server_url }} + GITHUB_REPOSITORY: ${{ github.repository }} + GITHUB_EVENT_NAME: ${{ inputs.trigger_event || github.event_name }} + run: | + chmod +x rvs_nightly_test.sh + mkdir -p ./reports + ./rvs_nightly_test.sh build-report + + - name: Upload report and logs + if: always() + uses: actions/upload-artifact@v4 + with: + name: rvs-nightly-docker-ubuntu24.04-report-${{ github.run_id }} + path: ./reports/ + retention-days: 30 + if-no-files-found: warn + + - name: Fail the workflow if level 4 failed + if: steps.report.outputs.overall == 'FAIL' || needs.run-rvs-level-4-in-docker.result == 'failure' + run: | + echo "::error::RVS docker (Ubuntu 24.04) test failure — level4 rc=${{ needs.run-rvs-level-4-in-docker.outputs.rc }}" + exit 1 diff --git a/.github/workflows/rvs-nightly-docker-ubuntu26.04-tests.yml b/.github/workflows/rvs-nightly-docker-ubuntu26.04-tests.yml new file mode 100644 index 000000000..fff6847a3 --- /dev/null +++ b/.github/workflows/rvs-nightly-docker-ubuntu26.04-tests.yml @@ -0,0 +1,346 @@ +name: RVS Nightly Tests (Docker Ubuntu 26.04) + +# Same flow as rvs-nightly-docker-ubuntu24.04-tests.yml but the ROCm runtime container uses ubuntu:26.04. +# ROCm SDK version is derived from the tar tarball name (-rMMmm.yyyymmdd-Linux.tar.gz). +# +# Default image delivery: build-on-target (docker build on the GPU node; no cross-host image transfer). +# Set workflow input build_on_target=false to fall back to scp/registry delivery from the runner. +# +# Image delivery (RVS_DOCKER_TRANSFER_MODE / workflow input docker_transfer_mode): +# auto (default) — build-on-target when build_on_target is true (default), else registry when +# RVS_DOCKER_REGISTRY is set, else scp (docker save | pigz/gzip | scp | docker load) +# scp — always save/scp/load (skip when target already has matching ROCM_VERSION) +# registry — docker push on build host, docker pull on target (requires RVS_DOCKER_REGISTRY) +# build-on-target — docker build on the GPU node (no cross-host image transfer) +# +# RVS tarball: downloaded on the orchestrator runner and scp'd to the target (not wget in container). +# +# Prerequisites: +# - Build host (self-hosted runner): curl/scp/ssh; HTTPS egress to tarball index; checkout supplies docker build context +# - Target GPU node: docker, HTTPS to nightly.repo.amd.com, /dev/kfd, /dev/dri, SSH user in docker group +# - ROCm SDK in docker image: therock-dist-linux-multiarch-.tar.gz on ubuntu:26.04 +# - Secrets: RVS_TARGET_NODE, RVS_TARGET_SSH_KEY (optional RVS_TARGET_USER, RVS_TARBALL_INDEX_URL) +# - Optional: vars.ROCM_SDK_NIGHTLY_BASE_URL to override the ROCm SDK nightly tarball host +# - Optional: vars.RVS_DOCKER_REGISTRY + secrets RVS_DOCKER_REGISTRY_USER/PASSWORD for registry mode +# +# See README_NIGHTLY_TESTS.md for SSH / tarball index configuration. + +on: + workflow_dispatch: + inputs: + trigger_event: + description: 'Label for report only (e.g. schedule or push); optional' + required: false + default: '' + tarball_url: + description: 'Override tarball URL (default: latest from index — secrets.RVS_TARBALL_INDEX_URL)' + required: false + default: '' + target_node: + description: 'GPU target hostname/IP (default: secrets.RVS_TARGET_NODE)' + required: false + default: '' + target_user: + description: 'SSH user on target (default: secrets.RVS_TARGET_USER)' + required: false + default: '' + docker_image: + description: 'ROCm runtime image (default: vars.RVS_NIGHTLY_DOCKER_UBUNTU26_IMAGE or rvs-nightly-rocm-ubuntu26.04:latest)' + required: false + default: '' + build_docker_image: + description: 'Build rvs-nightly-rocm-ubuntu26.04 on the build host before scp/registry delivery (only when build_on_target is false)' + required: false + type: boolean + default: false + build_on_target: + description: 'Build the ROCm docker image on the GPU target (default; skips cross-host image transfer)' + required: false + type: boolean + default: true + docker_transfer_mode: + description: 'Image delivery — auto, scp, registry, or build-on-target (default auto)' + required: false + default: 'auto' + remote_work_dir: + description: 'Work dir on target (default: /tmp/rvs-nightly-docker-ubuntu26.04-)' + required: false + default: '' + fallback_latest_sdk: + description: 'If the exact SDK date is missing, fall back to latest same line, then same major, then newest nightly' + required: false + type: boolean + default: true + +permissions: + contents: read + +concurrency: + group: rvs-nightly-docker-ubuntu26.04-${{ github.workflow }} + cancel-in-progress: false + +env: + TARBALL_INDEX_URL: ${{ secrets.RVS_TARBALL_INDEX_URL }} + RVS_NIGHTLY_DOCKER_IMAGE: ${{ inputs.docker_image || vars.RVS_NIGHTLY_DOCKER_UBUNTU26_IMAGE || 'rvs-nightly-rocm-ubuntu26.04:latest' }} + RVS_DOCKER_BUILD_DIR: .github/docker/rvs-nightly-rocm-ubuntu26.04 + RVS_DOCKER_IMAGE_ARCHIVE: rvs-nightly-rocm-ubuntu26.04-image.tar.gz + RVS_DOCKER_ON_TARGET: 'true' + RVS_DOCKER_TRANSFER_MODE: ${{ inputs.docker_transfer_mode || vars.RVS_DOCKER_TRANSFER_MODE || 'auto' }} + RVS_DOCKER_BUILD_ON_TARGET: ${{ (inputs.build_on_target != false) && (vars.RVS_DOCKER_BUILD_ON_TARGET != 'false') && 'true' || 'false' }} + RVS_DOCKER_REGISTRY: ${{ vars.RVS_DOCKER_REGISTRY || '' }} + RVS_DOCKER_REGISTRY_USER: ${{ secrets.RVS_DOCKER_REGISTRY_USER }} + RVS_DOCKER_REGISTRY_PASSWORD: ${{ secrets.RVS_DOCKER_REGISTRY_PASSWORD }} + RVS_DOCKER_SKIP_IF_PRESENT: 'true' + # Default true on schedule (inputs unset). github.event.inputs is a string; empty != 'false'. + # Opt out: uncheck workflow_dispatch fallback_latest_sdk, or set vars.RVS_DOCKER_SDK_FALLBACK_LATEST=false + RVS_DOCKER_SDK_FALLBACK_LATEST: ${{ (github.event.inputs.fallback_latest_sdk != 'false' && vars.RVS_DOCKER_SDK_FALLBACK_LATEST != 'false') && 'true' || 'false' }} + # ROCm SDK nightly host. Set vars.ROCM_SDK_NIGHTLY_BASE_URL (e.g. https://nightly.repo.amd.com/rocm/core/tarball). + # Optional vars.ROCM_SDK_NIGHTLY_INDEX_URL; if unset, BASE_URL is used for both listing and download. + ROCM_SDK_NIGHTLY_BASE_URL: ${{ vars.ROCM_SDK_NIGHTLY_BASE_URL || '' }} + ROCM_SDK_NIGHTLY_INDEX_URL: ${{ vars.ROCM_SDK_NIGHTLY_INDEX_URL || vars.ROCM_SDK_NIGHTLY_BASE_URL || '' }} + TARGET_ROCM_PATH: /opt/rocm/install + TARGET_NODE: ${{ inputs.target_node || secrets.RVS_TARGET_NODE }} + TARGET_USER: ${{ inputs.target_user || secrets.RVS_TARGET_USER }} + SSH_PRIVATE_KEY: ${{ secrets.RVS_TARGET_SSH_KEY }} + +jobs: + install-rvs-in-docker: + name: Ensure ROCm image (Ubuntu 26.04) on target and install RVS in docker + runs-on: ${{ vars.RVS_NIGHTLY_DOCKER_RUNNER_LABEL || vars.RVS_TEST_RUNNER_LABEL || 'self-hosted' }} + timeout-minutes: 180 + outputs: + tarball_url: ${{ steps.resolve.outputs.tarball_url }} + tarball_name: ${{ steps.resolve.outputs.tarball_name }} + remote_work_dir: ${{ steps.prepare.outputs.remote_work_dir }} + rocm_major: ${{ steps.prepare.outputs.rocm_major }} + install_dir: ${{ steps.prepare.outputs.install_dir }} + rvs_bin: ${{ steps.prepare.outputs.rvs_bin }} + env: + TARGET_ROCM_PATH: /opt/rocm/install + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + + - name: Resolve latest tarball URL + id: resolve + env: + INPUT_URL: ${{ inputs.tarball_url }} + run: | + set -euo pipefail + INDEX_URL="${TARBALL_INDEX_URL:-}" + if [ -z "${INPUT_URL:-}" ] && [ -z "${INDEX_URL}" ]; then + echo "::error::No tarball index URL (set RVS_TARBALL_INDEX_URL) and no tarball_url input." + exit 1 + fi + fetch_latest() { + local idx="$1" + local html raw name + html="$(curl -sSL --max-time 60 --retry 2 --retry-delay 2 "$idx" 2>/dev/null || true)" + raw="$(printf '%s' "$html" | grep -oE 'amdrocm[0-9]*-rvs-[0-9A-Za-z._\-]+-Linux\.tar\.gz' 2>/dev/null || true)" + name="$(printf '%s\n' "$raw" | sort -uV | tail -n 1 || true)" + [ -n "$name" ] || return 1 + printf '%s/%s\n' "${idx%/}" "$name" + } + if [ -n "${INPUT_URL:-}" ]; then + URL="$INPUT_URL" + else + URL="$(fetch_latest "$INDEX_URL")" + fi + [ -n "${URL:-}" ] || { echo "::error::Could not resolve tarball URL"; exit 1; } + NAME="$(basename "$URL" | tr -d '\r\n')" + if [[ "$NAME" != *-Linux.tar.gz ]]; then + echo "::error::Docker nightly tests require a *-Linux.tar.gz relocatable tarball; got: ${NAME}" >&2 + exit 1 + fi + if [[ ! "$NAME" =~ -r[0-9]{4}\.[0-9]{8}-Linux\.tar\.gz$ ]]; then + echo "::error::Tar tarball must include -rMMmm.yyyymmdd- (e.g. -r0715.20260724-Linux.tar.gz); got: ${NAME}" >&2 + exit 1 + fi + { + echo "tarball_url<> "$GITHUB_OUTPUT" + { + echo "TARBALL_URL<> "$GITHUB_ENV" + echo "Latest tarball : $NAME" + + - name: Validate configuration and derive paths + id: prepare + env: + TARBALL_NAME: ${{ steps.resolve.outputs.tarball_name }} + INPUT_REMOTE_WORK_DIR: ${{ inputs.remote_work_dir || format('/tmp/rvs-nightly-docker-ubuntu26.04-{0}', github.run_id) }} + VAR_REMOTE_WORK_DIR: ${{ vars.RVS_REMOTE_WORK_DIR }} + GITHUB_RUN_ID: ${{ github.run_id }} + run: | + chmod +x rvs_nightly_test.sh + if [ -z "${TARGET_NODE:-}" ]; then + echo "::error::No target node (set secrets.RVS_TARGET_NODE or workflow input target_node)." >&2 + exit 1 + fi + ./rvs_nightly_test.sh validate-config + + - name: Setup SSH to target node + env: + REMOTE_WORK_DIR: ${{ steps.prepare.outputs.remote_work_dir }} + run: ./rvs_nightly_test.sh setup-ssh + + - name: Export paths for remaining steps + env: + TARBALL_NAME: ${{ steps.resolve.outputs.tarball_name }} + run: | + { + echo "REMOTE_WORK_DIR=${{ steps.prepare.outputs.remote_work_dir }}" + echo "ROCM_MAJOR=${{ steps.prepare.outputs.rocm_major }}" + echo "INSTALL_DIR=${{ steps.prepare.outputs.install_dir }}" + echo "RVS_BIN=${{ steps.prepare.outputs.rvs_bin }}" + } >> "$GITHUB_ENV" + if [[ "$TARBALL_NAME" =~ -r([0-9]{2})([0-9]{2})\.([0-9]{8})-Linux\.tar\.gz$ ]]; then + echo "RVS_DOCKER_ROCM_VERSION=$((10#${BASH_REMATCH[1]})).$((10#${BASH_REMATCH[2]})).0a${BASH_REMATCH[3]}" >> "$GITHUB_ENV" + fi + + - name: Build ROCm docker image on build host + if: inputs.build_docker_image && inputs.build_on_target == false + env: + TARBALL_NAME: ${{ steps.resolve.outputs.tarball_name }} + run: | + chmod +x .github/docker/build-rocm-sdk-image.sh + fallback_args=() + if [ "${RVS_DOCKER_SDK_FALLBACK_LATEST}" = true ]; then + fallback_args=(--fallback-latest-sdk) + fi + ./.github/docker/build-rocm-sdk-image.sh --context .github/docker/rvs-nightly-rocm-ubuntu26.04 --from-tarball "$TARBALL_NAME" "${fallback_args[@]}" + + - name: Verify ROCm docker image on build host + if: inputs.build_docker_image && inputs.build_on_target == false + run: | + chmod +x rvs_nightly_docker.sh + ./rvs_nightly_docker.sh pull-image + + - name: Ensure ROCm docker image on GPU target + env: + TARBALL_NAME: ${{ steps.resolve.outputs.tarball_name }} + run: | + chmod +x rvs_nightly_docker.sh + ./rvs_nightly_docker.sh ensure-image-on-target + + - name: Verify ROCm in docker on target + run: ./rvs_nightly_docker.sh verify-rocm + + - name: Install RVS in docker on target + run: ./rvs_nightly_docker.sh install-rvs + + run-rvs-level-4-in-docker: + name: Run RVS level 4 in docker (Ubuntu 26.04) on target + needs: [install-rvs-in-docker] + runs-on: ${{ vars.RVS_NIGHTLY_DOCKER_RUNNER_LABEL || vars.RVS_TEST_RUNNER_LABEL || 'self-hosted' }} + timeout-minutes: 180 + outputs: + rc: ${{ steps.level4.outputs.rc }} + start: ${{ steps.level4.outputs.start }} + end: ${{ steps.level4.outputs.end }} + rvs_version: ${{ steps.versions.outputs.rvs_version }} + target_rocm_version: ${{ steps.versions.outputs.target_rocm_version }} + env: + TARGET_ROCM_PATH: /opt/rocm/install + TARGET_NODE: ${{ inputs.target_node || secrets.RVS_TARGET_NODE }} + TARGET_USER: ${{ inputs.target_user || secrets.RVS_TARGET_USER }} + SSH_PRIVATE_KEY: ${{ secrets.RVS_TARGET_SSH_KEY }} + RVS_DOCKER_ON_TARGET: 'true' + RVS_DOCKER_BUILD_DIR: .github/docker/rvs-nightly-rocm-ubuntu26.04 + RVS_DOCKER_IMAGE_ARCHIVE: rvs-nightly-rocm-ubuntu26.04-image.tar.gz + REMOTE_WORK_DIR: ${{ needs.install-rvs-in-docker.outputs.remote_work_dir }} + ROCM_MAJOR: ${{ needs.install-rvs-in-docker.outputs.rocm_major }} + INSTALL_DIR: ${{ needs.install-rvs-in-docker.outputs.install_dir }} + RVS_BIN: ${{ needs.install-rvs-in-docker.outputs.rvs_bin }} + RVS_NIGHTLY_DOCKER_IMAGE: ${{ inputs.docker_image || vars.RVS_NIGHTLY_DOCKER_UBUNTU26_IMAGE || 'rvs-nightly-rocm-ubuntu26.04:latest' }} + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + + - name: Setup SSH to target node + env: + REMOTE_WORK_DIR: ${{ needs.install-rvs-in-docker.outputs.remote_work_dir }} + run: | + chmod +x rvs_nightly_test.sh rvs_nightly_docker.sh + ./rvs_nightly_test.sh setup-ssh + + - name: Run RVS level 4 in docker on target + id: level4 + run: ./rvs_nightly_docker.sh run-level4 + + - name: Capture RVS and ROCm versions from docker on target + id: versions + if: always() + run: ./rvs_nightly_docker.sh capture-versions + + - name: Collect logs from target + if: always() + run: ./rvs_nightly_test.sh collect-logs + + - name: Upload test logs + if: always() + uses: actions/upload-artifact@v4 + with: + name: rvs-nightly-docker-ubuntu26.04-logs-${{ github.run_id }} + path: ./reports/ + retention-days: 30 + if-no-files-found: warn + + create-test-report: + name: Create Test Report + needs: [install-rvs-in-docker, run-rvs-level-4-in-docker] + runs-on: ${{ vars.RUNNER_LABEL_UTILITY || 'ubuntu-latest' }} + if: always() && !cancelled() && needs.install-rvs-in-docker.result == 'success' && needs.run-rvs-level-4-in-docker.result != 'skipped' + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + + - name: Download test logs + uses: actions/download-artifact@v4 + with: + name: rvs-nightly-docker-ubuntu26.04-logs-${{ github.run_id }} + path: ./reports + continue-on-error: true + + - name: Build test report + id: report + env: + TARBALL_URL: ${{ needs.install-rvs-in-docker.outputs.tarball_url }} + TARBALL_NAME: ${{ needs.install-rvs-in-docker.outputs.tarball_name }} + TARGET_ROCM_PATH: /opt/rocm/install + REMOTE_WORK_DIR: ${{ needs.install-rvs-in-docker.outputs.remote_work_dir }} + RVS_BIN: ${{ needs.install-rvs-in-docker.outputs.rvs_bin }} + RVS_LEVEL4_RC: ${{ needs.run-rvs-level-4-in-docker.outputs.rc }} + RVS_LEVEL4_START: ${{ needs.run-rvs-level-4-in-docker.outputs.start }} + RVS_LEVEL4_END: ${{ needs.run-rvs-level-4-in-docker.outputs.end }} + RVS_VERSION: ${{ needs.run-rvs-level-4-in-docker.outputs.rvs_version }} + TARGET_ROCM_VERSION: ${{ needs.run-rvs-level-4-in-docker.outputs.target_rocm_version }} + GITHUB_RUN_ID: ${{ github.run_id }} + GITHUB_SERVER_URL: ${{ github.server_url }} + GITHUB_REPOSITORY: ${{ github.repository }} + GITHUB_EVENT_NAME: ${{ inputs.trigger_event || github.event_name }} + run: | + chmod +x rvs_nightly_test.sh + mkdir -p ./reports + ./rvs_nightly_test.sh build-report + + - name: Upload report and logs + if: always() + uses: actions/upload-artifact@v4 + with: + name: rvs-nightly-docker-ubuntu26.04-report-${{ github.run_id }} + path: ./reports/ + retention-days: 30 + if-no-files-found: warn + + - name: Fail the workflow if level 4 failed + if: steps.report.outputs.overall == 'FAIL' || needs.run-rvs-level-4-in-docker.result == 'failure' + run: | + echo "::error::RVS docker (Ubuntu 26.04) test failure — level4 rc=${{ needs.run-rvs-level-4-in-docker.outputs.rc }}" + exit 1 diff --git a/.github/workflows/rvs-nightly-tests.yml b/.github/workflows/rvs-nightly-tests.yml new file mode 100644 index 000000000..9c48e8cc8 --- /dev/null +++ b/.github/workflows/rvs-nightly-tests.yml @@ -0,0 +1,340 @@ +name: RVS Nightly Tests + +# Picks up the latest RVS tarball from the tarball index (secrets.RVS_TARBALL_INDEX_URL +# into env.TARBALL_INDEX_URL below — no baked-in default URL), copies it +# to a configurable target node, installs it there, runs RVS level 4 on that node, +# and uploads a Markdown report + logs. +# +# Orchestration lives in this workflow; install/test/report steps run via +# rvs_nightly_test.sh (same pattern as build-relocatable-packages.yml + +# build_packages_local.sh). +# +# Required repo configuration: see README_NIGHTLY_TESTS.md +# +# Tarball index resolution, tarball download, and scp to the GPU target all run on +# the self-hosted orchestrator job below — the target never curls the index or CDN. +# +# Triggers +# -------- +# - workflow_dispatch: manual run, or dispatched by Build Relocatable Packages +# after a successful push to main/master or scheduled package build +# - trigger_event input (schedule|push) is set by the package-build dispatch step + +on: + workflow_dispatch: + inputs: + trigger_event: + description: 'Package-build event that triggered this run (schedule or push); empty for manual runs' + required: false + default: '' + tarball_url: + description: 'Override tarball URL (default: latest from index — secrets.RVS_TARBALL_INDEX_URL)' + required: false + default: '' + target_node: + description: 'Hostname/IP of node to install RVS on and run tests (default: secrets.RVS_TARGET_NODE)' + required: false + default: '' + target_user: + description: 'SSH user on the target node (default: secrets.RVS_TARGET_USER)' + required: false + default: '' + remote_work_dir: + description: 'Working dir on the target node (default: vars.RVS_REMOTE_WORK_DIR or /tmp/rvs-nightly-)' + required: false + default: '' + target_rocm_path: + description: 'ROCm tarball install root on the target node (default: vars.RVS_TARGET_ROCM_PATH)' + required: false + default: '' + +permissions: + contents: read + +concurrency: + group: rvs-nightly-${{ github.workflow }} + cancel-in-progress: false + +env: + # HTTPS directory listing for nightly tarballs — set repo secret RVS_TARBALL_INDEX_URL (e.g. https:///nightly/rvs/tar/). + TARBALL_INDEX_URL: ${{ secrets.RVS_TARBALL_INDEX_URL }} + +jobs: + install-rvs-on-target: + name: Install RVS on target node + runs-on: ${{ vars.RVS_NIGHTLY_INDEX_RUNNER_LABEL || vars.RVS_TEST_RUNNER_LABEL || 'self-hosted' }} + timeout-minutes: 120 + outputs: + tarball_url: ${{ steps.resolve.outputs.tarball_url }} + tarball_name: ${{ steps.resolve.outputs.tarball_name }} + remote_work_dir: ${{ steps.prepare.outputs.remote_work_dir }} + rocm_major: ${{ steps.prepare.outputs.rocm_major }} + install_dir: ${{ steps.prepare.outputs.install_dir }} + rvs_bin: ${{ steps.prepare.outputs.rvs_bin }} + env: + TARGET_NODE: ${{ inputs.target_node || secrets.RVS_TARGET_NODE }} + TARGET_USER: ${{ inputs.target_user || secrets.RVS_TARGET_USER }} + TARGET_ROCM_PATH: ${{ inputs.target_rocm_path || vars.RVS_TARGET_ROCM_PATH }} + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + + - name: Restore last-seen marker (scheduled build path) + if: inputs.trigger_event == 'schedule' + uses: actions/cache@v4 + with: + path: last_tarball_marker.txt + key: rvs-nightly-tarball-${{ github.run_id }} + restore-keys: | + rvs-nightly-tarball- + + - name: Resolve latest tarball URL (orchestrator only) + id: resolve + env: + # Do not set INDEX_URL here from ${{ env.* }} — it can be empty on some runners and + # masks workflow env. Use inherited TARBALL_INDEX_URL inside the script. + INPUT_URL: ${{ inputs.tarball_url }} + EVENT_NAME: ${{ github.event_name }} + WORKFLOW_RUN_EVENT: ${{ inputs.trigger_event }} + run: | + set -euo pipefail + + # Workflow-level env.TARBALL_INDEX_URL (from secret RVS_TARBALL_INDEX_URL). + INDEX_URL="${TARBALL_INDEX_URL:-}" + + if [ -z "${INPUT_URL:-}" ] && [ -z "${INDEX_URL}" ]; then + echo "::error::No tarball index URL (set Actions secret RVS_TARBALL_INDEX_URL) and no workflow_dispatch.tarball_url." + exit 1 + fi + + fetch_latest() { + local idx="$1" + local html raw name + html="$(curl -sSL --max-time 60 --retry 2 --retry-delay 2 "$idx" 2>/dev/null || true)" + if [ -z "$html" ]; then + echo "::error::curl returned empty body for index URL (check orchestrator HTTPS egress and URL): ${idx}" >&2 + return 0 + fi + # grep exits 1 when there are no matches — must not trip pipefail before we check results. + raw="$(printf '%s' "$html" | grep -oE 'amdrocm[0-9]*-rvs-[0-9A-Za-z._\-]+-Linux\.tar\.gz' 2>/dev/null || true)" + name="$(printf '%s\n' "$raw" | sort -uV | tail -n 1 || true)" + if [ -z "$name" ]; then + echo "::error::Index page contained no amdrocm*-rvs-*-Linux.tar.gz links (check listing format or regex). Index URL: ${idx}" >&2 + echo "::notice::First 800 chars of response (for debugging):" >&2 + printf '%.800s\n' "$html" >&2 || true + echo "" >&2 + return 0 + fi + printf '%s/%s\n' "${idx%/}" "$name" + } + + if [ -n "${INPUT_URL:-}" ]; then + URL="$INPUT_URL" + else + URL="$(fetch_latest "$INDEX_URL")" + fi + + if [ -z "${URL:-}" ]; then + echo "::error::Could not resolve a tarball URL from index: ${INDEX_URL}" + exit 1 + fi + + NAME="$(basename "$URL")" + + if [ "$WORKFLOW_RUN_EVENT" = "schedule" ] && [ -f last_tarball_marker.txt ]; then + LAST="$(cat last_tarball_marker.txt)" + if [ "$LAST" = "$NAME" ]; then + echo "::notice::Latest tarball ($NAME) unchanged since last scheduled package build — running tests anyway." + else + echo "::notice::New tarball (previous: $LAST, now: $NAME)." + fi + fi + echo "$NAME" > last_tarball_marker.txt + + # Use heredoc delimiter syntax so URLs with %, &, query strings, etc. are not + # truncated or dropped when propagated to other jobs via job outputs. + { + echo "tarball_url<> "$GITHUB_OUTPUT" + + # Later steps in this job read TARBALL_URL / TARBALL_NAME (download-tarball, etc.) + { + echo "TARBALL_URL<> "$GITHUB_ENV" + + echo "Latest tarball : $NAME" + echo "URL : $URL" + + - name: Save last-seen marker (scheduled build path) + if: inputs.trigger_event == 'schedule' + uses: actions/cache/save@v4 + with: + path: last_tarball_marker.txt + key: rvs-nightly-tarball-${{ steps.resolve.outputs.tarball_name }} + + - name: Validate configuration and derive paths + id: prepare + env: + TARBALL_NAME: ${{ steps.resolve.outputs.tarball_name }} + INPUT_REMOTE_WORK_DIR: ${{ inputs.remote_work_dir }} + VAR_REMOTE_WORK_DIR: ${{ vars.RVS_REMOTE_WORK_DIR }} + GITHUB_RUN_ID: ${{ github.run_id }} + run: | + chmod +x rvs_nightly_test.sh + ./rvs_nightly_test.sh validate-config + + - name: Export paths for remaining steps + run: | + { + echo "REMOTE_WORK_DIR=${{ steps.prepare.outputs.remote_work_dir }}" + echo "ROCM_MAJOR=${{ steps.prepare.outputs.rocm_major }}" + echo "INSTALL_DIR=${{ steps.prepare.outputs.install_dir }}" + echo "RVS_BIN=${{ steps.prepare.outputs.rvs_bin }}" + } >> "$GITHUB_ENV" + + - name: Setup SSH for target node + env: + SSH_PRIVATE_KEY: ${{ secrets.RVS_TARGET_SSH_KEY }} + run: | + chmod +x rvs_nightly_test.sh + ./rvs_nightly_test.sh setup-ssh + + - name: Download tarball on orchestrator runner + run: ./rvs_nightly_test.sh download-tarball + + - name: Copy tarball to target node + run: ./rvs_nightly_test.sh copy-to-target + + - name: Verify ROCm prerequisites on target node + run: ./rvs_nightly_test.sh verify-rocm + + - name: Install RVS on target node + run: ./rvs_nightly_test.sh install-rvs + + - name: Verify RVS binary library resolution on target node + run: ./rvs_nightly_test.sh verify-rvs-binary + + - name: Cleanup local SSH state + if: always() + run: ./rvs_nightly_test.sh cleanup-local-ssh + + run-rvs-level-4: + name: Run RVS level 4 on target node + needs: [install-rvs-on-target] + runs-on: ${{ vars.RVS_TEST_RUNNER_LABEL || 'self-hosted' }} + timeout-minutes: 180 + outputs: + rc: ${{ steps.level4.outputs.rc }} + start: ${{ steps.level4.outputs.start }} + end: ${{ steps.level4.outputs.end }} + rvs_version: ${{ steps.versions.outputs.rvs_version }} + target_rocm_version: ${{ steps.versions.outputs.target_rocm_version }} + env: + TARGET_NODE: ${{ inputs.target_node || secrets.RVS_TARGET_NODE }} + TARGET_USER: ${{ inputs.target_user || secrets.RVS_TARGET_USER }} + TARGET_ROCM_PATH: ${{ inputs.target_rocm_path || vars.RVS_TARGET_ROCM_PATH }} + REMOTE_WORK_DIR: ${{ needs.install-rvs-on-target.outputs.remote_work_dir }} + ROCM_MAJOR: ${{ needs.install-rvs-on-target.outputs.rocm_major }} + INSTALL_DIR: ${{ needs.install-rvs-on-target.outputs.install_dir }} + RVS_BIN: ${{ needs.install-rvs-on-target.outputs.rvs_bin }} + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + with: + submodules: recursive + + - name: Setup SSH for target node + env: + SSH_PRIVATE_KEY: ${{ secrets.RVS_TARGET_SSH_KEY }} + run: | + chmod +x rvs_nightly_test.sh + ./rvs_nightly_test.sh setup-ssh + + - name: Run RVS level 4 on target node + id: level4 + run: ./rvs_nightly_test.sh run-level4 + + - name: Collect logs from target node + if: always() + run: ./rvs_nightly_test.sh collect-logs + + - name: Capture RVS and ROCm versions from target + id: versions + if: always() + run: ./rvs_nightly_test.sh capture-versions + + - name: Upload test logs (intermediate) + if: always() + uses: actions/upload-artifact@v4 + with: + name: rvs-nightly-logs-${{ github.run_id }} + path: ./reports/ + retention-days: 30 + if-no-files-found: warn + + - name: Cleanup remote work dir and local SSH state + if: always() + run: | + ./rvs_nightly_test.sh cleanup-remote + ./rvs_nightly_test.sh cleanup-local-ssh + + create-test-report: + name: Create Test Report + needs: [install-rvs-on-target, run-rvs-level-4] + runs-on: ${{ vars.RUNNER_LABEL_UTILITY || 'ubuntu-latest' }} + if: always() && !cancelled() && needs.install-rvs-on-target.result == 'success' && needs.run-rvs-level-4.result != 'skipped' + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + with: + submodules: recursive + + - name: Download test logs + uses: actions/download-artifact@v4 + with: + name: rvs-nightly-logs-${{ github.run_id }} + path: ./reports + continue-on-error: true + + - name: Build test report + id: report + env: + TARBALL_URL: ${{ needs.install-rvs-on-target.outputs.tarball_url }} + TARBALL_NAME: ${{ needs.install-rvs-on-target.outputs.tarball_name }} + TARGET_ROCM_PATH: ${{ inputs.target_rocm_path || vars.RVS_TARGET_ROCM_PATH }} + REMOTE_WORK_DIR: ${{ needs.install-rvs-on-target.outputs.remote_work_dir }} + RVS_BIN: ${{ needs.install-rvs-on-target.outputs.rvs_bin }} + RVS_LEVEL4_RC: ${{ needs.run-rvs-level-4.outputs.rc }} + RVS_LEVEL4_START: ${{ needs.run-rvs-level-4.outputs.start }} + RVS_LEVEL4_END: ${{ needs.run-rvs-level-4.outputs.end }} + RVS_VERSION: ${{ needs.run-rvs-level-4.outputs.rvs_version }} + TARGET_ROCM_VERSION: ${{ needs.run-rvs-level-4.outputs.target_rocm_version }} + GITHUB_RUN_ID: ${{ github.run_id }} + GITHUB_SERVER_URL: ${{ github.server_url }} + GITHUB_REPOSITORY: ${{ github.repository }} + GITHUB_EVENT_NAME: ${{ inputs.trigger_event || github.event_name }} + run: | + chmod +x rvs_nightly_test.sh + mkdir -p ./reports + ./rvs_nightly_test.sh build-report + + - name: Upload report and logs + if: always() + uses: actions/upload-artifact@v4 + with: + name: rvs-nightly-report-${{ github.run_id }} + path: ./reports/ + retention-days: 30 + if-no-files-found: warn + + - name: Fail the workflow if level 4 failed + if: steps.report.outputs.overall == 'FAIL' || needs.run-rvs-level-4.result == 'failure' + run: | + echo "::error::RVS test failure — level4 rc=${{ needs.run-rvs-level-4.outputs.rc }}" + exit 1 diff --git a/.github/workflows/rvs-pr-tests.yml b/.github/workflows/rvs-pr-tests.yml new file mode 100644 index 000000000..cee46bfe6 --- /dev/null +++ b/.github/workflows/rvs-pr-tests.yml @@ -0,0 +1,352 @@ +name: RVS PR Tests + +# Runs after a successful "Build Relocatable Packages" run on pull_request (dispatched +# from that workflow with pr_number + build_run_number), or on workflow_dispatch. +# Post-build dispatch does not resolve the package URL — the install job on the +# self-hosted runner probes the manylinux_2_28 listing (curl, lab CDN egress). +# Manual runs: package_url or secret. See README_PR_TESTS.md. +# +# Required repo configuration: README_PR_TESTS.md (same secrets/vars as nightly for SSH/ROCm). + +on: + workflow_dispatch: + inputs: + pr_number: + description: 'PR number (set by Build Relocatable Packages post-build dispatch)' + required: false + default: '' + build_run_number: + description: 'Build workflow run number (post-build dispatch)' + required: false + default: '' + build_ref_name: + description: 'GITHUB_REF_NAME from the package build (/merge); set by post-build dispatch' + required: false + default: '' + package_url: + description: 'HTTPS …/rvs/ root, …/manylinux_*/ listing, direct *-Linux.tar.gz, or .deb (empty → secrets; not used when pr_number + build_run_number are set)' + required: false + default: '' + target_node: + description: 'Hostname/IP of node to install RVS on and run tests (default: secrets.RVS_TARGET_NODE)' + required: false + default: '' + target_user: + description: 'SSH user on the target node (default: secrets.RVS_TARGET_USER)' + required: false + default: '' + remote_work_dir: + description: 'Working dir on the target node (default: vars.RVS_REMOTE_WORK_DIR or /tmp/rvs-pr-)' + required: false + default: '' + target_rocm_path: + description: 'ROCm tarball install root on the target node (default: vars.RVS_TARGET_ROCM_PATH)' + required: false + default: '' + +permissions: + contents: read + +concurrency: + group: rvs-pr-${{ github.workflow }}-${{ github.run_id }} + cancel-in-progress: false + +env: + PR_DEB_PACKAGE_URL: ${{ secrets.RVS_PR_DEB_PACKAGE_URL || secrets.RVS_PR_PACKAGE_URL || '' }} + +jobs: + install-rvs-on-target: + name: Install RVS on target node + runs-on: ${{ vars.RVS_NIGHTLY_INDEX_RUNNER_LABEL || vars.RVS_TEST_RUNNER_LABEL || 'self-hosted' }} + timeout-minutes: 120 + outputs: + tarball_url: ${{ steps.resolve.outputs.tarball_url }} + tarball_name: ${{ steps.resolve.outputs.tarball_name }} + remote_work_dir: ${{ steps.prepare.outputs.remote_work_dir }} + rocm_major: ${{ steps.prepare.outputs.rocm_major }} + install_dir: ${{ steps.prepare.outputs.install_dir }} + rvs_bin: ${{ steps.prepare.outputs.rvs_bin }} + env: + TARGET_NODE: ${{ inputs.target_node || secrets.RVS_TARGET_NODE }} + TARGET_USER: ${{ inputs.target_user || secrets.RVS_TARGET_USER }} + TARGET_ROCM_PATH: ${{ inputs.target_rocm_path || vars.RVS_TARGET_ROCM_PATH }} + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + + - name: Set package URL (manual) + if: inputs.pr_number == '' || inputs.build_run_number == '' + env: + PACKAGE_INPUT: ${{ inputs.package_url }} + SECRET_PKG: ${{ secrets.RVS_PR_DEB_PACKAGE_URL || secrets.RVS_PR_PACKAGE_URL }} + run: | + set -euo pipefail + u="${PACKAGE_INPUT:-}" + if [ -z "${u}" ]; then + u="${SECRET_PKG:-}" + fi + if [ -z "${u}" ]; then + echo "::error::Set Actions secret RVS_PR_DEB_PACKAGE_URL (or RVS_PR_PACKAGE_URL), or pass workflow_dispatch input package_url." + exit 1 + fi + { + echo "PR_PACKAGE_ROOT_URL<> "$GITHUB_ENV" + + - name: Resolve manylinux listing (post-build dispatch, self-hosted) + if: inputs.pr_number != '' && inputs.build_run_number != '' + env: + SECRET_PKG: ${{ secrets.RVS_PR_DEB_PACKAGE_URL || secrets.RVS_PR_PACKAGE_URL }} + PR_NUMBER: ${{ inputs.pr_number }} + RUN_NUMBER: ${{ inputs.build_run_number }} + BUILD_REF_NAME: ${{ inputs.build_ref_name }} + run: | + set -euo pipefail + python3 <<'PY' + import os + import subprocess + import sys + + base = (os.environ.get("SECRET_PKG") or "").strip().rstrip("/") + if not base: + print("::error::Set RVS_PR_DEB_PACKAGE_URL (or RVS_PR_PACKAGE_URL) to the HTTPS rvs listing base (…/rvs) for post-build runs.", file=sys.stderr) + sys.exit(1) + low = base.lower() + if low.endswith(".deb") or low.endswith(".tar.gz"): + print("::error::For post-build dispatch, set the secret to the /rvs HTTPS base only, not a direct .deb or .tar.gz URL.", file=sys.stderr) + sys.exit(1) + + pr = int(os.environ["PR_NUMBER"]) + run_number = int(os.environ["RUN_NUMBER"]) + ref_name = (os.environ.get("BUILD_REF_NAME") or "").strip() + # Primary: exact rvs-s3-upload-route.sh layout (rvs/${GITHUB_REF_NAME}/${GITHUB_RUN_NUMBER}/manylinux_2_28) + candidates = [] + if ref_name: + candidates.append(f"{base}/{ref_name}/{run_number}/manylinux_2_28") + candidates.extend([ + f"{base}/{pr}/merge/{run_number}/manylinux_2_28", + f"{base}/{run_number}/merge/{pr}/manylinux_2_28", + ]) + + def listing_reachable(url: str) -> bool: + p = subprocess.run( + [ + "curl", + "-sfS", + "--max-time", + "60", + "-A", + "Mozilla/5.0 (compatible; RVS-PR-Tests/1.0)", + "-o", + "/dev/null", + url.rstrip("/") + "/", + ], + capture_output=True, + ) + return p.returncode == 0 + + listing = "" + for cand in candidates: + if listing_reachable(cand): + listing = cand + break + if not listing: + print( + "::error::Could not reach a manylinux_2_28 listing for this PR build. Tried:\n " + + "\n ".join(c + "/" for c in candidates), + file=sys.stderr, + ) + sys.exit(1) + print(f"::notice::Using manylinux_2_28 listing from this Build workflow run: {listing}/") + + out_path = os.environ["GITHUB_ENV"] + with open(out_path, "a", encoding="utf-8") as f: + f.write("PR_PACKAGE_ROOT_URL<> "$GITHUB_ENV" + + # Job outputs can drop tarball_url/name when passed to create-test-report; artifact is reliable. + - name: Upload package metadata for report job + uses: actions/upload-artifact@v4 + with: + name: rvs-pr-pkg-meta-${{ github.run_id }} + path: ./pkg-meta/ + retention-days: 1 + if-no-files-found: error + + - name: Setup SSH for target node + env: + SSH_PRIVATE_KEY: ${{ secrets.RVS_TARGET_SSH_KEY }} + run: | + chmod +x rvs_pr_test.sh + ./rvs_pr_test.sh setup-ssh + + - name: Download package on orchestrator runner + run: ./rvs_pr_test.sh download-tarball + + - name: Copy package to target node + run: ./rvs_pr_test.sh copy-to-target + + - name: Verify ROCm prerequisites on target node + run: ./rvs_pr_test.sh verify-rocm + + - name: Install RVS on target node + run: ./rvs_pr_test.sh install-rvs + + - name: Verify RVS binary library resolution on target node + run: ./rvs_pr_test.sh verify-rvs-binary + + - name: Cleanup local SSH state + if: always() + run: ./rvs_pr_test.sh cleanup-local-ssh + + run-rvs-level-4: + name: Run RVS level 4 on target node + needs: [install-rvs-on-target] + if: needs.install-rvs-on-target.result == 'success' + runs-on: ${{ vars.RVS_TEST_RUNNER_LABEL || 'self-hosted' }} + timeout-minutes: 180 + outputs: + rc: ${{ steps.level4.outputs.rc }} + start: ${{ steps.level4.outputs.start }} + end: ${{ steps.level4.outputs.end }} + rvs_version: ${{ steps.versions.outputs.rvs_version }} + target_rocm_version: ${{ steps.versions.outputs.target_rocm_version }} + env: + TARGET_NODE: ${{ inputs.target_node || secrets.RVS_TARGET_NODE }} + TARGET_USER: ${{ inputs.target_user || secrets.RVS_TARGET_USER }} + TARGET_ROCM_PATH: ${{ inputs.target_rocm_path || vars.RVS_TARGET_ROCM_PATH }} + REMOTE_WORK_DIR: ${{ needs.install-rvs-on-target.outputs.remote_work_dir }} + ROCM_MAJOR: ${{ needs.install-rvs-on-target.outputs.rocm_major }} + INSTALL_DIR: ${{ needs.install-rvs-on-target.outputs.install_dir }} + RVS_BIN: ${{ needs.install-rvs-on-target.outputs.rvs_bin }} + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + + - name: Setup SSH for target node + env: + SSH_PRIVATE_KEY: ${{ secrets.RVS_TARGET_SSH_KEY }} + run: | + chmod +x rvs_pr_test.sh + ./rvs_pr_test.sh setup-ssh + + - name: Run RVS level 4 on target node + id: level4 + run: ./rvs_pr_test.sh run-level4 + + - name: Collect logs from target node + if: always() + run: ./rvs_pr_test.sh collect-logs + + - name: Capture RVS and ROCm versions from target + id: versions + if: always() + run: ./rvs_pr_test.sh capture-versions + + - name: Upload test logs (intermediate) + if: always() + uses: actions/upload-artifact@v4 + with: + name: rvs-pr-logs-${{ github.run_id }} + path: ./reports/ + retention-days: 30 + if-no-files-found: warn + + - name: Cleanup remote work dir and local SSH state + if: always() + run: | + ./rvs_pr_test.sh cleanup-remote + ./rvs_pr_test.sh cleanup-local-ssh + + create-test-report: + name: Create Test Report + needs: [install-rvs-on-target, run-rvs-level-4] + runs-on: ${{ vars.RUNNER_LABEL_UTILITY || 'ubuntu-latest' }} + if: always() && !cancelled() && needs.install-rvs-on-target.result == 'success' && needs.run-rvs-level-4.result != 'skipped' + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + + - name: Download test logs + uses: actions/download-artifact@v4 + with: + name: rvs-pr-logs-${{ github.run_id }} + path: ./reports + continue-on-error: true + + - name: Download package metadata + uses: actions/download-artifact@v4 + with: + name: rvs-pr-pkg-meta-${{ github.run_id }} + path: ./pkg-meta + + - name: Build test report + id: report + env: + TARBALL_URL: ${{ needs.install-rvs-on-target.outputs.tarball_url }} + TARBALL_NAME: ${{ needs.install-rvs-on-target.outputs.tarball_name }} + TARGET_ROCM_PATH: ${{ inputs.target_rocm_path || vars.RVS_TARGET_ROCM_PATH }} + REMOTE_WORK_DIR: ${{ needs.install-rvs-on-target.outputs.remote_work_dir }} + RVS_BIN: ${{ needs.install-rvs-on-target.outputs.rvs_bin }} + RVS_LEVEL4_RC: ${{ needs.run-rvs-level-4.outputs.rc }} + RVS_LEVEL4_START: ${{ needs.run-rvs-level-4.outputs.start }} + RVS_LEVEL4_END: ${{ needs.run-rvs-level-4.outputs.end }} + RVS_VERSION: ${{ needs.run-rvs-level-4.outputs.rvs_version }} + TARGET_ROCM_VERSION: ${{ needs.run-rvs-level-4.outputs.target_rocm_version }} + GITHUB_RUN_ID: ${{ github.run_id }} + GITHUB_SERVER_URL: ${{ github.server_url }} + GITHUB_REPOSITORY: ${{ github.repository }} + GITHUB_EVENT_NAME: ${{ github.event_name }} + run: | + chmod +x rvs_pr_test.sh + mkdir -p ./reports + ./rvs_pr_test.sh build-report + + - name: Upload report and logs + if: always() + uses: actions/upload-artifact@v4 + with: + name: rvs-pr-report-${{ github.run_id }} + path: ./reports/ + retention-days: 30 + if-no-files-found: warn + + - name: Fail the workflow if level 4 failed + if: steps.report.outputs.overall == 'FAIL' || needs.run-rvs-level-4.result == 'failure' + run: | + echo "::error::RVS test failure — level4 rc=${{ needs.run-rvs-level-4.outputs.rc }}" + exit 1 diff --git a/.github/workflows/rvs-release-tests.yml b/.github/workflows/rvs-release-tests.yml new file mode 100644 index 000000000..d26d22bcd --- /dev/null +++ b/.github/workflows/rvs-release-tests.yml @@ -0,0 +1,295 @@ +name: RVS Release Tests + +# Picks up the latest RVS release tarball from the release CDN index, copies it +# to a configurable target node, installs it there, runs RVS level 5 on that node, +# and uploads a Markdown report + logs. +# +# Orchestration lives in this workflow; install/test/report steps run via +# rvs_release_test.sh (level 5; release tarball channel only). +# +# See README_RELEASE_TESTS.md for setup and prerequisites. +# +# Triggers +# -------- +# - workflow_dispatch: manual run, or dispatched by Build Relocatable Packages +# after a successful build on a release/* branch + +on: + workflow_dispatch: + inputs: + tarball_url: + description: 'Override tarball URL (default: latest from release index)' + required: false + default: '' + target_node: + description: 'Hostname/IP of node to install RVS on and run tests (default: secrets.RVS_RELEASE_TARGET_NODE)' + required: false + default: '' + target_user: + description: 'SSH user on the target node (default: secrets.RVS_RELEASE_TARGET_USER)' + required: false + default: '' + remote_work_dir: + description: 'Working dir on the target node (default: /tmp/rvs-release-)' + required: false + default: '' + target_rocm_path: + description: 'ROCm tarball install root on the target node (default: secrets.RVS_RELEASE_TARGET_ROCM_PATH)' + required: false + default: '' + +permissions: + contents: read + +concurrency: + group: rvs-release-${{ github.workflow }} + cancel-in-progress: false + +env: + # HTTPS directory listing for release tarballs — set repo secret RVS_RELEASE_TARBALL_INDEX_URL. + TARBALL_INDEX_URL: ${{ secrets.RVS_RELEASE_TARBALL_INDEX_URL }} + +jobs: + install-rvs-on-target: + name: Install RVS on target node + runs-on: ${{ vars.RVS_RELEASE_INDEX_RUNNER_LABEL || vars.RVS_NIGHTLY_INDEX_RUNNER_LABEL || vars.RVS_TEST_RUNNER_LABEL || 'self-hosted' }} + timeout-minutes: 120 + outputs: + tarball_url: ${{ steps.resolve.outputs.tarball_url }} + tarball_name: ${{ steps.resolve.outputs.tarball_name }} + remote_work_dir: ${{ steps.prepare.outputs.remote_work_dir }} + rocm_major: ${{ steps.prepare.outputs.rocm_major }} + install_dir: ${{ steps.prepare.outputs.install_dir }} + rvs_bin: ${{ steps.prepare.outputs.rvs_bin }} + env: + TARGET_NODE: ${{ inputs.target_node || secrets.RVS_RELEASE_TARGET_NODE }} + TARGET_USER: ${{ inputs.target_user || secrets.RVS_RELEASE_TARGET_USER }} + TARGET_ROCM_PATH: ${{ inputs.target_rocm_path || secrets.RVS_RELEASE_TARGET_ROCM_PATH }} + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + + - name: Resolve latest release tarball URL (orchestrator only) + id: resolve + env: + INPUT_URL: ${{ inputs.tarball_url }} + run: | + set -euo pipefail + + INDEX_URL="${TARBALL_INDEX_URL:-}" + + if [ -z "${INPUT_URL:-}" ] && [ -z "${INDEX_URL}" ]; then + echo "::error::No tarball index URL (set Actions secret RVS_RELEASE_TARBALL_INDEX_URL) and no workflow_dispatch.tarball_url." + exit 1 + fi + + fetch_latest() { + local idx="$1" + local html raw name + html="$(curl -sSL --max-time 60 --retry 2 --retry-delay 2 "$idx" 2>/dev/null || true)" + if [ -z "$html" ]; then + echo "::error::curl returned empty body for index URL (check orchestrator HTTPS egress and URL): ${idx}" >&2 + return 0 + fi + raw="$(printf '%s' "$html" | grep -oE 'amdrocm[0-9]*-rvs-[0-9A-Za-z._\-]+-Linux\.tar\.gz' 2>/dev/null || true)" + name="$(printf '%s\n' "$raw" | sort -uV | tail -n 1 || true)" + if [ -z "$name" ]; then + echo "::error::Index page contained no amdrocm*-rvs-*-Linux.tar.gz links (check listing format or regex). Index URL: ${idx}" >&2 + echo "::notice::First 800 chars of response (for debugging):" >&2 + printf '%.800s\n' "$html" >&2 || true + echo "" >&2 + return 0 + fi + printf '%s/%s\n' "${idx%/}" "$name" + } + + if [ -n "${INPUT_URL:-}" ]; then + URL="$INPUT_URL" + else + URL="$(fetch_latest "$INDEX_URL")" + fi + + if [ -z "${URL:-}" ]; then + echo "::error::Could not resolve a tarball URL from index: ${INDEX_URL}" + exit 1 + fi + + NAME="$(basename "$URL")" + + { + echo "tarball_url<> "$GITHUB_OUTPUT" + + { + echo "TARBALL_URL<> "$GITHUB_ENV" + + echo "Latest release tarball : $NAME" + echo "URL : $URL" + + - name: Validate configuration and derive paths + id: prepare + env: + TARBALL_NAME: ${{ steps.resolve.outputs.tarball_name }} + INPUT_REMOTE_WORK_DIR: ${{ inputs.remote_work_dir }} + VAR_REMOTE_WORK_DIR: ${{ vars.RVS_RELEASE_REMOTE_WORK_DIR }} + GITHUB_RUN_ID: ${{ github.run_id }} + run: | + chmod +x rvs_release_test.sh + ./rvs_release_test.sh validate-config + + - name: Export paths for remaining steps + run: | + { + echo "REMOTE_WORK_DIR=${{ steps.prepare.outputs.remote_work_dir }}" + echo "ROCM_MAJOR=${{ steps.prepare.outputs.rocm_major }}" + echo "INSTALL_DIR=${{ steps.prepare.outputs.install_dir }}" + echo "RVS_BIN=${{ steps.prepare.outputs.rvs_bin }}" + } >> "$GITHUB_ENV" + + - name: Setup SSH for target node + env: + SSH_PRIVATE_KEY: ${{ secrets.RVS_RELEASE_TARGET_SSH_KEY }} + run: | + chmod +x rvs_release_test.sh + ./rvs_release_test.sh setup-ssh + + - name: Download tarball on orchestrator runner + run: ./rvs_release_test.sh download-tarball + + - name: Copy tarball to target node + run: ./rvs_release_test.sh copy-to-target + + - name: Verify ROCm prerequisites on target node + run: ./rvs_release_test.sh verify-rocm + + - name: Install RVS on target node + run: ./rvs_release_test.sh install-rvs + + - name: Verify RVS binary library resolution on target node + run: ./rvs_release_test.sh verify-rvs-binary + + - name: Cleanup local SSH state + if: always() + run: ./rvs_release_test.sh cleanup-local-ssh + + run-rvs-level-5: + name: Run RVS level 5 on target node + needs: [install-rvs-on-target] + runs-on: ${{ vars.RVS_RELEASE_TEST_RUNNER_LABEL || vars.RVS_TEST_RUNNER_LABEL || 'self-hosted' }} + timeout-minutes: 480 + outputs: + rc: ${{ steps.level5.outputs.rc }} + start: ${{ steps.level5.outputs.start }} + end: ${{ steps.level5.outputs.end }} + rvs_version: ${{ steps.versions.outputs.rvs_version }} + target_rocm_version: ${{ steps.versions.outputs.target_rocm_version }} + env: + TARGET_NODE: ${{ inputs.target_node || secrets.RVS_RELEASE_TARGET_NODE }} + TARGET_USER: ${{ inputs.target_user || secrets.RVS_RELEASE_TARGET_USER }} + TARGET_ROCM_PATH: ${{ inputs.target_rocm_path || secrets.RVS_RELEASE_TARGET_ROCM_PATH }} + REMOTE_WORK_DIR: ${{ needs.install-rvs-on-target.outputs.remote_work_dir }} + ROCM_MAJOR: ${{ needs.install-rvs-on-target.outputs.rocm_major }} + INSTALL_DIR: ${{ needs.install-rvs-on-target.outputs.install_dir }} + RVS_BIN: ${{ needs.install-rvs-on-target.outputs.rvs_bin }} + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + with: + submodules: recursive + + - name: Setup SSH for target node + env: + SSH_PRIVATE_KEY: ${{ secrets.RVS_RELEASE_TARGET_SSH_KEY }} + run: | + chmod +x rvs_release_test.sh + ./rvs_release_test.sh setup-ssh + + - name: Run RVS level 5 on target node + id: level5 + run: ./rvs_release_test.sh run-level5 + + - name: Collect logs from target node + if: always() + run: ./rvs_release_test.sh collect-logs + + - name: Capture RVS and ROCm versions from target + id: versions + if: always() + run: ./rvs_release_test.sh capture-versions + + - name: Upload test logs (intermediate) + if: always() + uses: actions/upload-artifact@v4 + with: + name: rvs-release-logs-${{ github.run_id }} + path: ./reports/ + retention-days: 30 + if-no-files-found: warn + + - name: Cleanup remote work dir and local SSH state + if: always() + run: | + ./rvs_release_test.sh cleanup-remote + ./rvs_release_test.sh cleanup-local-ssh + + create-test-report: + name: Create Test Report + needs: [install-rvs-on-target, run-rvs-level-5] + runs-on: ${{ vars.RUNNER_LABEL_UTILITY || 'ubuntu-latest' }} + if: always() && !cancelled() && needs.install-rvs-on-target.result == 'success' && needs.run-rvs-level-5.result != 'skipped' + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + with: + submodules: recursive + + - name: Download test logs + uses: actions/download-artifact@v4 + with: + name: rvs-release-logs-${{ github.run_id }} + path: ./reports + continue-on-error: true + + - name: Build test report + id: report + env: + TARBALL_URL: ${{ needs.install-rvs-on-target.outputs.tarball_url }} + TARBALL_NAME: ${{ needs.install-rvs-on-target.outputs.tarball_name }} + TARGET_ROCM_PATH: ${{ inputs.target_rocm_path || secrets.RVS_RELEASE_TARGET_ROCM_PATH }} + REMOTE_WORK_DIR: ${{ needs.install-rvs-on-target.outputs.remote_work_dir }} + RVS_BIN: ${{ needs.install-rvs-on-target.outputs.rvs_bin }} + RVS_LEVEL5_RC: ${{ needs.run-rvs-level-5.outputs.rc }} + RVS_LEVEL5_START: ${{ needs.run-rvs-level-5.outputs.start }} + RVS_LEVEL5_END: ${{ needs.run-rvs-level-5.outputs.end }} + RVS_VERSION: ${{ needs.run-rvs-level-5.outputs.rvs_version }} + TARGET_ROCM_VERSION: ${{ needs.run-rvs-level-5.outputs.target_rocm_version }} + GITHUB_RUN_ID: ${{ github.run_id }} + GITHUB_SERVER_URL: ${{ github.server_url }} + GITHUB_REPOSITORY: ${{ github.repository }} + GITHUB_EVENT_NAME: ${{ github.event_name }} + run: | + chmod +x rvs_release_test.sh + mkdir -p ./reports + ./rvs_release_test.sh build-report + + - name: Upload report and logs + if: always() + uses: actions/upload-artifact@v4 + with: + name: rvs-release-report-${{ github.run_id }} + path: ./reports/ + retention-days: 30 + if-no-files-found: warn + + - name: Fail the workflow if level 5 failed + if: steps.report.outputs.overall == 'FAIL' || needs.run-rvs-level-5.result == 'failure' + run: | + echo "::error::RVS release test failure — level5 rc=${{ needs.run-rvs-level-5.outputs.rc }}" + exit 1 diff --git a/.github/workflows/unsigned-release-candidate-promotion.yml b/.github/workflows/unsigned-release-candidate-promotion.yml new file mode 100644 index 000000000..c76a61a45 --- /dev/null +++ b/.github/workflows/unsigned-release-candidate-promotion.yml @@ -0,0 +1,457 @@ +name: Unsigned Release Candidate Promotion + +# Copies packages from release/rvs/{deb,rpm,tar}/ to release/unsigned/packages/{deb,rpm}/ +# and release/unsigned/tarball/, mirroring the layout and repodata structure used by +# nightly/unsigned/. +# +# Release packages are named with the GitHub run number of the build job as the +# release segment, e.g.: +# amdrocm7-rvs_1.3.15-12345_amd64.deb +# amdrocm7-rvs-1.3.15-12345..x86_64.rpm +# amdrocm7-rvs-1.3.15-12345-Linux.tar.gz +# +# The run_number input selects exactly those three files. + +on: + workflow_dispatch: + inputs: + run_number: + description: > + GitHub Actions run number of the build-relocatable-packages workflow run that + produced the release packages to promote (e.g. "12345"). This number is + embedded as the release segment in every package filename: + amdrocm7-rvs-1.3.15-12345..x86_64.rpm. Exactly one .deb, one .rpm, and one + .tar.gz must match; the step fails if zero or more than one file is found. + required: true + type: string + +permissions: + contents: read + id-token: write + +jobs: + promote-release-unsigned: + name: Promote release/rvs → release/unsigned (run ${{ github.event.inputs.run_number }}) + runs-on: ${{ vars.RUNNER_LABEL_UTILITY || 'ubuntu-latest' }} + + env: + RUN_NUMBER: ${{ github.event.inputs.run_number }} + AWS_S3_BUCKET: ${{ vars.AWS_S3_BUCKET }} + + steps: + - name: Checkout Repository + uses: actions/checkout@v4 + + # ── AWS setup ──────────────────────────────────────────────────────────── + - name: Install AWS CLI + run: | + if command -v aws >/dev/null 2>&1; then + aws --version + exit 0 + fi + python3 -m pip install --user --break-system-packages awscli \ + || python3 -m pip install --user awscli + echo "$HOME/.local/bin" >> "$GITHUB_PATH" + export PATH="$HOME/.local/bin:$PATH" + aws --version + + - name: Configure AWS credentials + run: | + TOKEN=$(curl -s -H "Authorization: bearer $ACTIONS_ID_TOKEN_REQUEST_TOKEN" \ + "${ACTIONS_ID_TOKEN_REQUEST_URL}&audience=sts.amazonaws.com" \ + | python3 -c "import sys,json; print(json.load(sys.stdin)['value'])") + CREDS=$(aws sts assume-role-with-web-identity \ + --role-arn "${{ secrets.AWS_ROLE_ARN }}" \ + --role-session-name "github-actions-${{ github.run_id }}" \ + --web-identity-token "$TOKEN" \ + --query 'Credentials' --output json) + echo "AWS_ACCESS_KEY_ID=$(echo "$CREDS" | python3 -c "import sys,json; print(json.load(sys.stdin)['AccessKeyId'])")" >> "$GITHUB_ENV" + echo "AWS_SECRET_ACCESS_KEY=$(echo "$CREDS" | python3 -c "import sys,json; print(json.load(sys.stdin)['SecretAccessKey'])")" >> "$GITHUB_ENV" + echo "AWS_SESSION_TOKEN=$(echo "$CREDS" | python3 -c "import sys,json; print(json.load(sys.stdin)['SessionToken'])")" >> "$GITHUB_ENV" + echo "AWS_DEFAULT_REGION=us-east-1" >> "$GITHUB_ENV" + echo "::add-mask::$(echo "$CREDS" | python3 -c "import sys,json; print(json.load(sys.stdin)['SecretAccessKey'])")" + echo "::add-mask::$(echo "$CREDS" | python3 -c "import sys,json; print(json.load(sys.stdin)['SessionToken'])")" + env: + AWS_DEFAULT_REGION: us-east-1 + + # ── Validate input ─────────────────────────────────────────────────────── + - name: Validate inputs + run: | + if [ -z "${RUN_NUMBER}" ]; then + echo "::error::run_number input is required." >&2 + exit 1 + fi + if [ -z "${AWS_S3_BUCKET}" ]; then + echo "::error::AWS_S3_BUCKET repository variable is not set." >&2 + exit 1 + fi + echo "Promoting packages for build run number: ${RUN_NUMBER}" + echo "Source: s3://${AWS_S3_BUCKET}/release/rvs/{deb,rpm,tar}/" + echo "Destination: s3://${AWS_S3_BUCKET}/release/unsigned/{packages/deb,packages/rpm,tarball}/" + + # ── Install packaging tools ────────────────────────────────────────────── + - name: Install packaging tools (reprepro, createrepo_c) + run: | + export DEBIAN_FRONTEND=noninteractive + sudo apt-get update -qq + sudo apt-get install -y --no-install-recommends \ + reprepro \ + dpkg-dev \ + createrepo-c || \ + sudo apt-get install -y --no-install-recommends \ + reprepro \ + dpkg-dev + + # ── DEB: copy matching files + rebuild APT archive ─────────────────────── + - name: Copy DEB packages and rebuild APT archive + id: promote-deb + run: | + set -euo pipefail + BUCKET="${AWS_S3_BUCKET}" + SRC_DEB="release/rvs/deb" + DST_DEB="release/unsigned/packages/deb" + + STAGING=$(mktemp -d) + trap 'rm -rf "$STAGING"' EXIT + + # Download existing unsigned DEB archive (accumulate mode — no --delete). + # aws s3 sync exits 0 for a nonexistent prefix; || true is not needed + # and would mask real errors such as permission failures. + echo "Syncing existing ${DST_DEB} archive ..." + aws s3 sync "s3://${BUCKET}/${DST_DEB}/conf/" "${STAGING}/conf/" --no-progress + aws s3 sync "s3://${BUCKET}/${DST_DEB}/pool/" "${STAGING}/pool/" --no-progress + aws s3 sync "s3://${BUCKET}/${DST_DEB}/dists/" "${STAGING}/dists/" --no-progress + + # Bootstrap reprepro distributions config if absent + if [ ! -f "${STAGING}/conf/distributions" ]; then + mkdir -p "${STAGING}/conf" + cat > "${STAGING}/conf/distributions" <<'DISTEOF' + Origin: ROCm Validation Suite + Label: stable + Suite: stable + Codename: stable + Architectures: amd64 + Components: main + Description: RVS unsigned release + DISTEOF + fi + + # List source DEBs and download the one matching the run number. + # Release DEBs are named amdrocm-rvs_-_amd64.deb. + # Match on the "-_" delimiter so run 123 cannot match 1234. + # Write to a temp file so a non-zero exit from aws is not hidden inside + # a process substitution and propagates correctly under set -euo pipefail. + INCOMING=$(mktemp -d) + aws s3 ls "s3://${BUCKET}/${SRC_DEB}/" --recursive \ + | awk '{print $NF}' > "${STAGING}/src_deb_keys.txt" + DEB_COPIED=0 + while IFS= read -r key; do + fname=$(basename "$key") + case "$fname" in amdrocm*-rvs*.deb) : ;; *) continue ;; esac + case "$fname" in *"-${RUN_NUMBER}_"*) : ;; *) continue ;; esac + echo " Downloading ${fname} ..." + aws s3 cp "s3://${BUCKET}/${key}" "${INCOMING}/${fname}" --no-progress + DEB_COPIED=$((DEB_COPIED + 1)) + done < "${STAGING}/src_deb_keys.txt" + + echo "deb_copied=${DEB_COPIED}" >> "$GITHUB_OUTPUT" + + if [ "${DEB_COPIED}" -eq 0 ]; then + echo "::error::No .deb files for run number '${RUN_NUMBER}' in s3://${BUCKET}/${SRC_DEB}/." >&2 + exit 1 + fi + + echo "Including ${DEB_COPIED} DEB(s) into reprepro archive ..." + for deb_file in "${INCOMING}"/amdrocm*-rvs*.deb; do + [ -f "$deb_file" ] || continue + pkg=$(dpkg-deb -f "$deb_file" Package) + ver=$(dpkg-deb -f "$deb_file" Version) + # Idempotent: remove same Package+Version before re-include + if reprepro -b "${STAGING}" listfilter stable \ + "Package (== ${pkg}), Version (== ${ver})" 2>/dev/null | grep -q .; then + reprepro -b "${STAGING}" -T deb removefilter stable \ + "Package (== ${pkg}), Version (== ${ver})" + fi + reprepro -b "${STAGING}" includedeb stable "$deb_file" + done + + echo "Syncing updated DEB archive to s3://${BUCKET}/${DST_DEB}/ ..." + aws s3 sync "${STAGING}/conf/" "s3://${BUCKET}/${DST_DEB}/conf/" --no-progress + aws s3 sync "${STAGING}/pool/" "s3://${BUCKET}/${DST_DEB}/pool/" --no-progress + aws s3 sync "${STAGING}/dists/" "s3://${BUCKET}/${DST_DEB}/dists/" --no-progress + echo "=== DEB archive updated at s3://${BUCKET}/${DST_DEB}/ ===" + aws s3 ls "s3://${BUCKET}/${DST_DEB}/dists/stable/" --human-readable + + # ── RPM: copy matching files + rebuild YUM repodata ───────────────────── + - name: Copy RPM packages and rebuild YUM repodata + id: promote-rpm + run: | + set -euo pipefail + BUCKET="${AWS_S3_BUCKET}" + SRC_RPM="release/rvs/rpm" + DST_RPM="release/unsigned/packages/rpm" + + STAGING=$(mktemp -d) + trap 'rm -rf "$STAGING"' EXIT + + # Packages live under an x86_64/ subdirectory so signing CI and + # yum/dnf clients can use release/unsigned/packages/rpm/x86_64/ as the baseurl. + mkdir -p "${STAGING}/x86_64" + + # Download existing unsigned RPMs (no repodata — will regenerate). + # aws s3 sync exits 0 for a nonexistent prefix; || true is not needed. + echo "Syncing existing ${DST_RPM}/x86_64/ RPMs ..." + aws s3 sync "s3://${BUCKET}/${DST_RPM}/x86_64/" "${STAGING}/x86_64/" \ + --exclude "repodata/*" --no-progress + + # List source RPMs into a temp file so aws exit status is not swallowed + # by a process substitution. + # Release RPMs are named amdrocm-rvs--..x86_64.rpm + # where is the RPM dist tag (e.g. el8 for manylinux_2_28). + # Match on "-." so run 123 cannot match 1234 + # (the "." matches the start of "..x86_64"). + aws s3 ls "s3://${BUCKET}/${SRC_RPM}/" --recursive \ + | awk '{print $NF}' > "${STAGING}/src_rpm_keys.txt" + RPM_COPIED=0 + while IFS= read -r key; do + fname=$(basename "$key") + case "$fname" in amdrocm*-rvs*.rpm) : ;; *) continue ;; esac + case "$fname" in *"-${RUN_NUMBER}."*) : ;; *) continue ;; esac + echo " Downloading ${fname} ..." + aws s3 cp "s3://${BUCKET}/${key}" "${STAGING}/x86_64/${fname}" --no-progress + RPM_COPIED=$((RPM_COPIED + 1)) + done < "${STAGING}/src_rpm_keys.txt" + + echo "rpm_copied=${RPM_COPIED}" >> "$GITHUB_OUTPUT" + + if [ "${RPM_COPIED}" -eq 0 ]; then + echo "::error::No .rpm files for run number '${RUN_NUMBER}' in s3://${BUCKET}/${SRC_RPM}/." >&2 + exit 1 + fi + if [ "${RPM_COPIED}" -gt 1 ]; then + echo "::error::${RPM_COPIED} .rpm files match run number '${RUN_NUMBER}'; expected exactly one." >&2 + exit 1 + fi + + # Compute SHA-256 of the promoted RPM and expose as a step output so + # the latest.json step can include it without re-downloading the file. + RPM_BASENAME=$(basename "${STAGING}/x86_64/"amdrocm*-rvs*.rpm) + RPM_SHA256=$(sha256sum "${STAGING}/x86_64/${RPM_BASENAME}" | awk '{print $1}') + echo "rpm_fname=${RPM_BASENAME}" >> "$GITHUB_OUTPUT" + echo "rpm_sha256=${RPM_SHA256}" >> "$GITHUB_OUTPUT" + + echo "Rebuilding RPM repodata under x86_64/ ..." + createrepo_c --simple-md-filenames --no-database --compress-type gz "${STAGING}/x86_64" \ + || createrepo "${STAGING}/x86_64" + + echo "Syncing updated RPM repo to s3://${BUCKET}/${DST_RPM}/x86_64/ ..." + aws s3 sync "${STAGING}/x86_64/" "s3://${BUCKET}/${DST_RPM}/x86_64/" --no-progress + echo "=== RPM repo updated at s3://${BUCKET}/${DST_RPM}/x86_64/ ===" + aws s3 ls "s3://${BUCKET}/${DST_RPM}/x86_64/repodata/" --human-readable + + # ── TAR: copy matching files + generate missing .sha256 sidecars ───────── + - name: Copy TAR packages + id: promote-tar + run: | + set -euo pipefail + BUCKET="${AWS_S3_BUCKET}" + SRC_TAR="release/rvs/tar" + DST_TAR="release/unsigned/tarball" + + STAGING=$(mktemp -d) + trap 'rm -rf "$STAGING"' EXIT + + # List source TARs into a temp file so aws exit status is not swallowed + # by a process substitution. + # Release TARs are named amdrocm-rvs---Linux.tar.gz. + # Match on "--Linux" so run 123 cannot match 1234. + # The same delimiter also matches the .sha256 sidecar. + aws s3 ls "s3://${BUCKET}/${SRC_TAR}/" --recursive \ + | awk '{print $NF}' > "${STAGING}/src_tar_keys.txt" + TAR_COPIED=0 + while IFS= read -r key; do + fname=$(basename "$key") + # Copy both .tar.gz and any existing .sha256 sidecars + case "$fname" in + amdrocm*-rvs*.tar.gz | amdrocm*-rvs*.tar.gz.sha256) : ;; + *) continue ;; + esac + case "$fname" in *"-${RUN_NUMBER}-Linux"*) : ;; *) continue ;; esac + echo " Downloading ${fname} ..." + aws s3 cp "s3://${BUCKET}/${key}" "${STAGING}/${fname}" --no-progress + case "$fname" in *.tar.gz) TAR_COPIED=$((TAR_COPIED + 1)) ;; esac + done < "${STAGING}/src_tar_keys.txt" + + echo "tar_copied=${TAR_COPIED}" >> "$GITHUB_OUTPUT" + + if [ "${TAR_COPIED}" -eq 0 ]; then + echo "::error::No .tar.gz files for run number '${RUN_NUMBER}' in s3://${BUCKET}/${SRC_TAR}/." >&2 + exit 1 + fi + if [ "${TAR_COPIED}" -gt 1 ]; then + echo "::error::${TAR_COPIED} .tar.gz files match run number '${RUN_NUMBER}'; expected exactly one." >&2 + exit 1 + fi + + # Generate .sha256 sidecar if the source did not include one. + # Run sha256sum from within STAGING so the path recorded in the sidecar + # is the bare filename, not the runner temp path (/tmp/…/file → file). + # This matches what sha256sum -c expects on the signing host. + TAR_BASENAME=$(basename "${STAGING}/"amdrocm*-rvs*.tar.gz) + SIDECAR="${STAGING}/${TAR_BASENAME}.sha256" + if [ ! -f "${SIDECAR}" ]; then + echo " Generating SHA-256 sidecar for ${TAR_BASENAME} ..." + (cd "${STAGING}" && sha256sum "${TAR_BASENAME}") > "${SIDECAR}" + fi + TAR_SHA256=$(awk '{print $1}' "${SIDECAR}") + echo "tar_fname=${TAR_BASENAME}" >> "$GITHUB_OUTPUT" + echo "tar_sha256=${TAR_SHA256}" >> "$GITHUB_OUTPUT" + + echo "Copying TARs to s3://${BUCKET}/${DST_TAR}/ ..." + aws s3 cp "${STAGING}/" "s3://${BUCKET}/${DST_TAR}/" \ + --recursive \ + --exclude "*" \ + --include "amdrocm*-rvs*.tar.gz" \ + --include "amdrocm*-rvs*.tar.gz.sha256" \ + --no-progress + echo "=== TAR packages updated at s3://${BUCKET}/${DST_TAR}/ ===" + aws s3 ls "s3://${BUCKET}/${DST_TAR}/" --human-readable + + # ── Publish release/unsigned/latest.json ───────────────────────────────── + - name: Publish release/unsigned/latest.json + run: | + set -euo pipefail + BUCKET="${AWS_S3_BUCKET}" + DST_DEB="release/unsigned/packages/deb" + DST_RPM="release/unsigned/packages/rpm" + DST_TAR="release/unsigned/tarball" + + # All three formats are required. The individual promote steps already + # fail on 0 or >1 matches and compute the SHA-256 there; expose them + # here via :? so a missing output is caught immediately. + : "${RPM_FNAME:?promote-rpm step output rpm_fname is missing}" + : "${RPM_SHA256:?promote-rpm step output rpm_sha256 is missing}" + : "${TAR_FNAME:?promote-tar step output tar_fname is missing}" + : "${TAR_SHA256:?promote-tar step output tar_sha256 is missing}" + + # Export WORKDIR so the embedded Python script inherits it + WORKDIR=$(mktemp -d) + export WORKDIR + trap 'rm -rf "$WORKDIR"' EXIT + + # Fetch the reprepro Packages index to extract DEB pool paths. + # This is a hard requirement — the DEB promote step must have succeeded. + DEB_STAGE="${WORKDIR}/deb" + mkdir -p "${DEB_STAGE}" + echo "Fetching Packages index for DEB metadata ..." + aws s3 cp \ + "s3://${BUCKET}/${DST_DEB}/dists/stable/main/binary-amd64/Packages" \ + "${DEB_STAGE}/Packages" --no-progress + + python3 - <> "$GITHUB_STEP_SUMMARY" diff --git a/.gitmodules b/.gitmodules index e69de29bb..8f5ec053a 100644 --- a/.gitmodules +++ b/.gitmodules @@ -0,0 +1,3 @@ +[submodule "external/TransferBench"] + path = external/TransferBench + url = https://github.com/ROCm/TransferBench.git diff --git a/CHANGELOG.md b/CHANGELOG.md index 5afd6ffa5..4b2ba8812 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,103 @@ Full documentation for RVS is available at [ROCmValidationSuite.Readme](https://github.com/ROCm/ROCmValidationSuite). +## RVS 1.7.0 + +### Added + +- Level-based test configurations (`-r` 1–5) for Radeon platforms: nv21, nv31, nv32, gfx1200, gfx1201, RX9060, RX9070, RX9070GRE, R9600D. +- PCI device ID auto-detection for Radeon GPUs in `gpu_get_platform_name()`. + +### Changed + +### Removed + +- Deprecated modules **GM**, **SMQT**, and **PESM** (deprecated since Nov 2024). GPU listing (`-g` / `--listGpus`) no longer depends on PESM. +- Unused **PERF** module (`perf.so`). +- Residual **EDP** sources (`edp.so`; product path removed earlier). + +## RVS 1.6.0 + +### Added + +- Test-level configurations for MI350P-450W and MI350P-600W. +- Support for new platform: MI450X. +- Support for Babel sustained bandwidth mode. +- Support for test duration selection using the `-t` option. +- Added iterations-based GST ramp-up support. + +### Changed + +- Updated and improved user-guide. +- Updated Babel JSON schema. +- Updated TransferBench to v1.69.00. + +## RVS 1.5.0 + +### Added + +- Bundled TransferBench as a git submodule under `external/TransferBench` and ship the TransferBench CLI in the same DEB/RPM as `rvs`. The CLI is provided for compatibility with existing TransferBench-based workflows; new work should use RVS (via the `pebb`/`pbqt` modules) or the TransferBench API directly. Opt out of the CLI build with `-DBUILD_TRANSFERBENCH_CLI=OFF`. +- Added the pulse stressor module (`pulse.so`) for GPU power pulse stress testing. **(Beta — not intended for production use.)** +- Added support for MI350X QPX mode. +- Added support for the MXFP8 data type. +- Added iterations-based GST hot run support. + +### Changed + +- Updated Babel output to report throughput instead of execution time. + +## RVS 1.4.0 + +### Added + +- TransferBench support to PEBB and PBQT. +- IET power stress test for MI250X. +- Babel support for normal distribution–based data initialization. +- Configurable Non-Temporal support in Babel. +- Support for action selection using the `-s` option. +- Support for module-based test execution using the `-m` option. +- Support for time‑duration based Babel test execution. +- MI350P configs for GST, IET and babel tests. + +## RVS 1.3.0 for ROCm 7.2 + +### Added + +- Babel subtest configurability + +## RVS 1.3.0 for ROCm 7.1.1 + +### Added + +- Support for different test levels with `-r` option for MI3XXX. +- Set compute type for DGEMM operations in MI350X and MI355X. + +## RVS 1.3.0 for ROCm 7.1.0 + +### Added + +- Added test summary and system overview. +- NPS2/DPX partition mode support for MI350X and MI355X. +- Support for Azure Linux and Alibaba Linux. +- Individual subtest configurability in Babel. + +## RVS 1.2.0 for ROCm 7.0.2 + +### Added + +- Support for Amazon Linux. + +### Changed + +- Update gst conf. files for MI350X and MI355X. +- Change GEMM execution method during ramp-up. + +## RVS 1.2.0 for ROCm 7.0.1 + +### Added + +- Support for new platform: RX9060 + ## RVS 1.2.0 for ROCm 7.0.0 ### Added @@ -100,3 +197,4 @@ Full documentation for RVS is available at [ROCmValidationSuite.Readme](https:// ### Optimized - In GST and IET modules, use of callback mechanism instead of polling for HIP stream reduced the CPU utilization %. + diff --git a/CMakeGtestDownload.cmake b/CMakeGtestDownload.cmake index d772419d6..344692fa5 100644 --- a/CMakeGtestDownload.cmake +++ b/CMakeGtestDownload.cmake @@ -23,14 +23,15 @@ ## ################################################################################ -cmake_minimum_required(VERSION 2.8.2) +cmake_minimum_required(VERSION 3.5.0) project(googletest-download NONE) include(ExternalProject) ExternalProject_Add(googletest GIT_REPOSITORY https://github.com/google/googletest.git - GIT_TAG v1.16.0 + GIT_TAG "6910c9d9165801d8827d628cb72eb7ea9dd538c5" #v1.16.0 + GIT_CONFIG "http.sslVerify=true" # Forces certificate validation SOURCE_DIR "${CMAKE_BINARY_DIR}/googletest-src" BINARY_DIR "${CMAKE_BINARY_DIR}/googletest-build" CONFIGURE_COMMAND "" diff --git a/CMakeLists.txt b/CMakeLists.txt index b16537f88..9b0e1c1bb 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -1,6 +1,6 @@ ################################################################################ ## -## Copyright (c) 2018-2025 Advanced Micro Devices, Inc. All rights reserved. +## Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. ## ## MIT LICENSE: ## Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -24,11 +24,15 @@ ################################################################################ cmake_minimum_required ( VERSION 3.5.0 ) -set(CMAKE_SHARED_LINKER_FLAGS_INIT "-Wl,--enable-new-dtags,--rpath,$ORIGIN:$ORIGIN/.." CACHE STRING "RUNPATH for libraries") -set(CMAKE_EXE_LINKER_FLAGS_INIT "-Wl,--enable-new-dtags,--rpath,$ORIGIN/../lib" CACHE STRING "RUNPATH for executables") +# Prefer RUNPATH (DT_RUNPATH); do not embed $ORIGIN here — CMAKE_BUILD_RPATH / CMAKE_INSTALL_RPATH below are the single source of truth. +set(CMAKE_SHARED_LINKER_FLAGS_INIT "-Wl,--enable-new-dtags" CACHE STRING "Shared linker flags") +set(CMAKE_EXE_LINKER_FLAGS_INIT "-Wl,--enable-new-dtags" CACHE STRING "Executable linker flags") +set( COMP_TYPE "applications" ) +set( BUILD_ENABLE_LINTIAN_OVERRIDES ON CACHE BOOL "Enable/Disable Lintian Overrides" ) +set( BUILD_DEBIAN_PKGING_FLAG ON CACHE BOOL "Internal Status Flag to indicate Debian Packaging Build" ) project ("rocm-validation-suite" - VERSION 1.2.0) + VERSION 1.7.0) # Default libdir to "lib", this skips GNUInstallDirs from trying to take a guess if it's unset: set(CMAKE_INSTALL_LIBDIR "lib" CACHE STRING "Library install directory") @@ -44,11 +48,12 @@ if ( ${CMAKE_BINARY_DIR} STREQUAL ${CMAKE_CURRENT_SOURCE_DIR}) message(FATAL "In-source build is not allowed") endif () enable_testing() -# In future use the api provided by rocm-core library to get ROCm Install path -# For backward compatibility use ROCM_PATH provided by build arguments. -# Using the api option is turned OFF by default +# When ON, link rocm-core: getROCmInstallPath for the ROCm install path, and +# getROCmVersion (rocm_version.h) for the YAML system-overview ROCm string — no +# .info file read for that path. When OFF, use ROCM_PATH and the version file +# under that prefix only. option(FETCH_ROCMPATH_FROM_ROCMCORE - "Use the api provided by rocm-core library to get ROCM_PATH" OFF) + "Use rocm-core C API for ROCm path and YAML system-overview ROCm version" OFF) # Prerequisite - Check if rocblas was already installed find_package (rocblas) @@ -142,28 +147,99 @@ else() #If hipblas-common not found message(FATAL_ERROR "hipblas-common not found !!! Install hipblas-common to proceed ...") endif(hipblas-common_FOUND) +# ROCm major version for package naming and install paths (per RFC0012) +set(ROCM_MAJOR_VERSION "0" CACHE STRING "ROCm major version (e.g. 7 for ROCm 7.x)") + +# Auto-compute patch version: count commits since last v..* tag +find_program(GIT_EXECUTABLE git) +if(GIT_EXECUTABLE) + # When cmake runs under sudo (e.g. Ubuntu CI), git refuses to operate in a + # directory owned by a different user. Mark the source dir as safe. + execute_process( + COMMAND ${GIT_EXECUTABLE} config --global --get-all safe.directory + OUTPUT_VARIABLE _SAFE_DIRS + OUTPUT_STRIP_TRAILING_WHITESPACE + ERROR_QUIET + ) + string(FIND "${_SAFE_DIRS}" "${CMAKE_SOURCE_DIR}" _SAFE_DIR_IDX) + if(_SAFE_DIR_IDX EQUAL -1) + execute_process( + COMMAND ${GIT_EXECUTABLE} config --global --add safe.directory "${CMAKE_SOURCE_DIR}" + ERROR_QUIET + ) + endif() + set(_TAG_PATTERN "v${PROJECT_VERSION_MAJOR}.${PROJECT_VERSION_MINOR}.*") + execute_process( + COMMAND ${GIT_EXECUTABLE} describe --tags --match "${_TAG_PATTERN}" --long + WORKING_DIRECTORY ${CMAKE_SOURCE_DIR} + OUTPUT_VARIABLE _GIT_DESCRIBE_OUTPUT + OUTPUT_STRIP_TRAILING_WHITESPACE + RESULT_VARIABLE _GIT_DESCRIBE_RESULT + ERROR_QUIET + ) + if(_GIT_DESCRIBE_RESULT EQUAL 0) + string(REGEX MATCH "-([0-9]+)-g[0-9a-f]+$" _MATCH "${_GIT_DESCRIBE_OUTPUT}") + set(RVS_PATCH_FROM_TAG "${CMAKE_MATCH_1}") + message(STATUS "RVS patch from tag: ${RVS_PATCH_FROM_TAG} (${_GIT_DESCRIBE_OUTPUT})") + else() + set(RVS_PATCH_FROM_TAG "0") + message(STATUS "No matching v${PROJECT_VERSION_MAJOR}.${PROJECT_VERSION_MINOR}.* tag found, patch defaults to 0") + endif() +else() + set(RVS_PATCH_FROM_TAG "0") + message(STATUS "git not found, patch version defaults to 0") +endif() + # Making ROCM_PATH, CMAKE_INSTALL_PREFIX, CPACK_PACKAGING_INSTALL_PREFIX as CACHE # variables since we will pass them as cmake params appropriately, and # all find_packages relevant to this build will be in ROCM path hence appending it to CMAKE_PREFIX_PATH set(ROCM_PATH "/opt/rocm" CACHE PATH "ROCM install path") -set(CMAKE_INSTALL_PREFIX "/opt/rocm" CACHE PATH "CMAKE installation directory") -set(CPACK_PACKAGING_INSTALL_PREFIX "/opt/rocm" CACHE PATH "Prefix used in built packages") +set(CMAKE_INSTALL_PREFIX "/opt/rocm/extras-${ROCM_MAJOR_VERSION}" CACHE PATH "CMAKE installation directory") +set(CPACK_PACKAGING_INSTALL_PREFIX "/opt/rocm/extras-${ROCM_MAJOR_VERSION}" CACHE PATH "Prefix used in built packages") list(APPEND CMAKE_PREFIX_PATH "${ROCM_PATH}") +# RPATH (build tree + install): canonical list in cmake_modules/RVSPackagedRpath.cmake. +# Packaged DEB/RPM/TGZ artifacts are normalized again via CPACK_PRE_BUILD_SCRIPTS. +set(CMAKE_SKIP_RPATH FALSE CACHE BOOL "Skip RPATH for build targets") +set(CMAKE_INSTALL_RPATH_USE_LINK_PATH FALSE CACHE BOOL "Append link paths to install RPATH") +include(${CMAKE_CURRENT_SOURCE_DIR}/cmake_modules/RVSPackagedRpath.cmake) +rvs_get_packaged_rpath_list(_RVS_INSTALL_RPATH) +set(CMAKE_INSTALL_RPATH "${_RVS_INSTALL_RPATH}" + CACHE STRING "RPATH for installed RVS binaries and modules") +set(CMAKE_BUILD_RPATH "${CMAKE_INSTALL_RPATH}") + +# CI (GitHub Actions): GITHUB_ACTIONS flags below need CMake 3.9+ / 3.16+ at configure time. +if("$ENV{GITHUB_ACTIONS}" STREQUAL "true") + message(STATUS "RVS RPATH (GitHub Actions): CMake ${CMAKE_VERSION}") + if(CMAKE_VERSION VERSION_GREATER_EQUAL "3.9") + set(CMAKE_SKIP_BUILD_RPATH TRUE CACHE BOOL "On GitHub Actions, skip CMake build-tree RPATH") + message(STATUS "RVS RPATH (GitHub Actions): CMAKE_SKIP_BUILD_RPATH=${CMAKE_SKIP_BUILD_RPATH}") + else() + message(STATUS "RVS RPATH (GitHub Actions): CMAKE_SKIP_BUILD_RPATH not applied (need CMake 3.9+)") + endif() + if(CMAKE_VERSION VERSION_GREATER_EQUAL "3.16") + set(CMAKE_INSTALL_REMOVE_ENVIRONMENT_RPATH TRUE CACHE BOOL "On GitHub Actions, strip implicit link-dir RPATH on install") + message(STATUS "RVS RPATH (GitHub Actions): CMAKE_INSTALL_REMOVE_ENVIRONMENT_RPATH=${CMAKE_INSTALL_REMOVE_ENVIRONMENT_RPATH}") + else() + message(STATUS "RVS RPATH (GitHub Actions): CMAKE_INSTALL_REMOVE_ENVIRONMENT_RPATH not applied (need CMake 3.16+)") + endif() +endif() + +# Compile-time default used by rvs_get_rocm_install_path_string() after +# getROCmInstallPath (if FETCH_*) and $ROCM_PATH env are considered. add_definitions(-DROCM_PATH="${ROCM_PATH}") if(FETCH_ROCMPATH_FROM_ROCMCORE) add_compile_options(-DFETCH_ROCMPATH_FROM_ROCMCORE=${FETCH_ROCMPATH_FROM_ROCMCORE}) - # rocm-core library will provide the ROCM_PATH during runtime - # Link the library as required set(ROCM_CORE "rocm-core") - # The absolute library path will be ${ROCM_PATH}/${RVS_LIB_PATH} - # Usage is ROCM_PATH will be fetched from rocm-core and append ${RVS_LIB_PATH} - add_definitions(-DRVS_LIB_PATH="${CMAKE_INSTALL_LIBDIR}/rvs") else() - add_definitions(-DRVS_LIB_PATH="${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_LIBDIR}/rvs") - # Set ROCM_CORE to empty, so no linking to rocm-core library/package set(ROCM_CORE "") endif() +# RVS module .so and shared data (conf) live under the RVS install prefix, not under ROCm. +add_definitions(-DRVS_LIB_PATH="${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_LIBDIR}/rvs") +add_definitions(-DRVS_DATA_ROOT="${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_DATADIR}/${CMAKE_PROJECT_NAME}") +# Suffixes relative to that prefix for runtime resolution from the rvs binary path. +add_definitions(-DRVS_RELPATH_DATA_DIR="${CMAKE_INSTALL_DATADIR}/${CMAKE_PROJECT_NAME}") +add_definitions(-DRVS_RELPATH_MODULE_LIB_DIR="${CMAKE_INSTALL_LIBDIR}/rvs") # # If the user specifies -DCMAKE_BUILD_TYPE on the command line, take their # definition and dump it in the cache along with proper documentation, @@ -207,54 +283,58 @@ else() endif() message(STATUS "RVS_OS_TYPE_NUM: ${RVS_OS_TYPE_NUM}") -## Set default module path if not already set -if ( NOT DEFINED CMAKE_MODULE_PATH ) - set ( CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/cmake_modules/" ) -endif() +## Always make our cmake_modules/ searchable: the find_package() calls above may have +## already populated CMAKE_MODULE_PATH (ROCm/TheRock configs do), which would otherwise +## leave include(utils) unresolved. Prepend so our modules win over same-named ones. +set ( CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/cmake_modules/" ${CMAKE_MODULE_PATH} ) +list ( REMOVE_DUPLICATES CMAKE_MODULE_PATH ) + ## Include common cmake modules include ( utils ) -## Update RVS version -set ( RVS_VERSION "${PROJECT_VERSION}" ) -get_version ( ${RVS_VERSION} ) +# get_version() in cmake_modules/utils.cmake: +# - Takes ONE string argument: a baseline semver (here: project(VERSION), e.g. "1.4.0"). +# - Runs parse_version() and sets VERSION_MAJOR/MINOR/PATCH (and related) for legacy uses +# (doxygen RVSVER, etc.). It does NOT set RVS_VERSION and does not know about git tag distance. +# - May override VERSION_PATCH from env ROCM_LIBPATCH_VERSION (ROCm xxyy encoding) — unrelated to RVS display semver. +get_version(${PROJECT_VERSION}) # Package Generator ####################################################### -set(CPACK_PACKAGE_NAME "rocm-validation-suite") +set(PKG_MAINTAINER_NM "ROCM Validation Suite Support") +set(PKG_MAINTAINER_EMAIL "rocm-validation-suite.support@amd.com") +set(CPACK_PACKAGE_NAME "rocm-validation-suite" CACHE STRING "Package name for CPack") +set(CMAKE_INSTALL_DOCDIR "${CMAKE_INSTALL_DATAROOTDIR}/doc/${CPACK_PACKAGE_NAME}" CACHE PATH "Documentation install directory" FORCE) set(CPACK_PACKAGE_DESCRIPTION "ROCm Validation Suite") set(CPACK_PACKAGE_DESCRIPTION_SUMMARY "Tool for validating AMD GPU functionality, performance, reliability in ROCm.") + +# CPack + "rvs -v" string: major/minor from project(), patch from git describe (RVS_PATCH_FROM_TAG) when > 0. set(CPACK_PACKAGE_VERSION_MAJOR "${PROJECT_VERSION_MAJOR}") set(CPACK_PACKAGE_VERSION_MINOR "${PROJECT_VERSION_MINOR}") -set(CPACK_PACKAGE_VERSION_PATCH "${PROJECT_VERSION_PATCH}") +if(RVS_PATCH_FROM_TAG GREATER "0") + set(CPACK_PACKAGE_VERSION_PATCH "${RVS_PATCH_FROM_TAG}") +else() + set(CPACK_PACKAGE_VERSION_PATCH "${PROJECT_VERSION_PATCH}") +endif() +set(RVS_VERSION "${CPACK_PACKAGE_VERSION_MAJOR}.${CPACK_PACKAGE_VERSION_MINOR}.${CPACK_PACKAGE_VERSION_PATCH}") +message(STATUS "RVS_VERSION (rvs -v / CPack, not get_version input): ${RVS_VERSION}") + set(CPACK_PACKAGE_VENDOR "Advanced Micro Devices Inc.") -set(CPACK_PACKAGE_CONTACT "ROCM Validation Suite Support ") +set(CPACK_PACKAGE_CONTACT "${PKG_MAINTAINER_NM} <${PKG_MAINTAINER_EMAIL}>") -# Package dependencies +# Package dependencies: support both legacy (ROCm <=6.x) and new amdrocm-* naming (TheRock/ROCm 8.x per RFC0009) +# DEB uses "old | new" OR syntax, RPM uses "(old or new)" boolean dependency syntax +# rocm-llvm: ROCm LLVM (includes libomp used by HIP/RVS); pull OpenMP runtime from the ROCm stack, not host libgomp. +set(ROCM_DEB_DEPS "hip-runtime-amd | amdrocm-runtime, comgr | amdrocm-runtime, hsa-rocr | amdrocm-runtime, rocblas | amdrocm-blas, amd-smi-lib | amdrocm-amdsmi, hiprand | amdrocm-rand, hipblaslt | amdrocm-blas, rocm-llvm | amdrocm-llvm") +set(ROCM_RPM_DEPS "(hip-runtime-amd or amdrocm-runtime), (comgr or amdrocm-runtime), (hsa-rocr or amdrocm-runtime), (rocblas or amdrocm-blas), (amd-smi-lib or amdrocm-amdsmi), (hiprand or amdrocm-rand), (hipblaslt or amdrocm-blas), (rocm-llvm or amdrocm-llvm)") if(FETCH_ROCMPATH_FROM_ROCMCORE) - set(ROCM_DEPENDENT_PACKAGES "hip-runtime-amd, comgr, hsa-rocr, rocblas, amd-smi-lib, hiprand, hipblaslt, ${ROCM_CORE}") -else() - set(ROCM_DEPENDENT_PACKAGES "hip-runtime-amd, comgr, hsa-rocr, rocblas, amd-smi-lib, hiprand, hipblaslt") -endif() -if (${RVS_OS_TYPE} STREQUAL "ubuntu") -set(CPACK_DEBIAN_PACKAGE_DEPENDS "${ROCM_DEPENDENT_PACKAGES}, libpci3, libyaml-cpp-dev") -elseif (${RVS_OS_TYPE} STREQUAL "sles") -set(CPACK_RPM_PACKAGE_REQUIRES "${ROCM_DEPENDENT_PACKAGES}, libpci3, libyaml-cpp0_6") -else () -# other supported rpm distros - RHEL, CentOS -set(CPACK_RPM_PACKAGE_REQUIRES "${ROCM_DEPENDENT_PACKAGES}, pciutils-libs, yaml-cpp") + set(ROCM_DEB_DEPS "${ROCM_DEB_DEPS}, ${ROCM_CORE} | amdrocm-base") + set(ROCM_RPM_DEPS "${ROCM_RPM_DEPS}, (${ROCM_CORE} or amdrocm-base)") endif() set(CPACK_ARCHIVE_COMPONENT_INSTALL ON) set(CPACK_COMPONENTS_ALL applications rvsmodule) -# Set default ROCm version for package naming -set(ROCM_VERSION_FOR_PACKAGE "99999") - -# Update ROCm version for package naming -if(DEFINED ENV{ROCM_LIBPATCH_VERSION}) - set(ROCM_VERSION_FOR_PACKAGE $ENV{ROCM_LIBPATCH_VERSION}) -endif() - # debian related changes if (DEFINED ENV{CPACK_DEBIAN_PACKAGE_RELEASE}) set(CPACK_DEBIAN_PACKAGE_RELEASE $ENV{CPACK_DEBIAN_PACKAGE_RELEASE}) @@ -275,12 +355,15 @@ if(CPACK_RPM_PACKAGE_RELEASE) set(CPACK_RPM_PACKAGE_RELEASE_DIST ON) endif() -# Complete RVS package version naming -set(CPACK_PACKAGE_VERSION "${CPACK_PACKAGE_VERSION_MAJOR}.${CPACK_PACKAGE_VERSION_MINOR}.${CPACK_PACKAGE_VERSION_PATCH}.${ROCM_VERSION_FOR_PACKAGE}") +set(CPACK_PACKAGE_VERSION "${CPACK_PACKAGE_VERSION_MAJOR}.${CPACK_PACKAGE_VERSION_MINOR}.${CPACK_PACKAGE_VERSION_PATCH}") set(CPACK_DEBIAN_FILE_NAME "DEB-DEFAULT") set(CPACK_RPM_FILE_NAME "RPM-DEFAULT") +# TGZ (STGZ): same version + release scheme as DEB/RPM — ---Linux.tar.gz +# CPACK_RPM_PACKAGE_RELEASE is set from the build script (e.g. r0711.20260423 or PR suffix) +set(CPACK_PACKAGE_FILE_NAME "${CPACK_PACKAGE_NAME}-${CPACK_PACKAGE_VERSION}-${CPACK_RPM_PACKAGE_RELEASE}-Linux" CACHE STRING "STGZ/CPack output file stem" FORCE) + # Set and Install license file set(CPACK_RESOURCE_FILE_LICENSE "${CMAKE_CURRENT_SOURCE_DIR}/LICENSE") install(FILES ${CPACK_RESOURCE_FILE_LICENSE} DESTINATION ${CPACK_PACKAGING_INSTALL_PREFIX}/${CMAKE_INSTALL_DATADIR}/doc/${CPACK_PACKAGE_NAME}) @@ -330,12 +413,100 @@ message (STATUS "CPACK_GENERATOR ${CPACK_GENERATOR}" ) ################################################################################ # check yaml-cpp available at configure time -find_package(yaml-cpp) -if (yaml-cpp_FOUND) - message("yaml-cpp found") +# First try to find static yaml-cpp library +find_library(YAML_CPP_STATIC_LIB + NAMES libyaml-cpp.a yaml-cpp.a + PATHS + /usr/lib/x86_64-linux-gnu + /usr/lib64 + /usr/lib + /usr/local/lib64 + /usr/local/lib + NO_DEFAULT_PATH # avoid searches in cmake_install_prefix etc + DOC "Path to yaml-cpp static library" +) + +find_path(YAML_CPP_INCLUDE_DIR + NAMES yaml-cpp/yaml.h + PATHS /usr/include /usr/local/include ${CMAKE_PREFIX_PATH}/include + DOC "Path to yaml-cpp include directory" +) + +# Set variables for yaml-cpp linking +set(YAML_CPP_STATIC_FOUND FALSE) +set(YAML_CPP_SHARED_FOUND FALSE) + +if (YAML_CPP_STATIC_LIB AND YAML_CPP_INCLUDE_DIR) + set(YAML_CPP_STATIC_FOUND TRUE) + message(STATUS "yaml-cpp static library found: ${YAML_CPP_STATIC_LIB}") + set(YAML_CPP_LIBRARIES ${YAML_CPP_STATIC_LIB}) + get_filename_component(YAML_CPP_LIBRARY_DIR ${YAML_CPP_STATIC_LIB} DIRECTORY) + + # Create imported target for static linking + add_library(yaml-cpp-static STATIC IMPORTED) + set_target_properties(yaml-cpp-static PROPERTIES + IMPORTED_LOCATION ${YAML_CPP_STATIC_LIB} + INTERFACE_INCLUDE_DIRECTORIES ${YAML_CPP_INCLUDE_DIR} + ) + + + # Use static library as the target + set(YAML_CPP_TARGET yaml-cpp-static) + else() - message(FATAL_ERROR "yaml-cpp not found !!! Install to proceed ...") -endif(yaml-cpp_FOUND) + # Fall back to find_package for shared library + find_package(yaml-cpp QUIET) + if (yaml-cpp_FOUND) + set(YAML_CPP_SHARED_FOUND TRUE) + message(STATUS "yaml-cpp shared library found: ${YAML_CPP_LIBRARIES}") + + if (TARGET yaml-cpp::yaml-cpp) + set(YAML_CPP_TARGET yaml-cpp::yaml-cpp) + get_target_property(YAML_CPP_INCLUDE_DIR yaml-cpp::yaml-cpp INTERFACE_INCLUDE_DIRECTORIES) + if (NOT YAML_CPP_INCLUDE_DIR) + set(YAML_CPP_INCLUDE_DIR ${YAML_CPP_INCLUDE_DIRS}) + endif() + else() + # Legacy variables from find_package + set(YAML_CPP_TARGET ${YAML_CPP_LIBRARIES}) + endif() + else() + message(FATAL_ERROR "yaml-cpp not found !!! Install yaml-cpp (preferably static version) to proceed ...") + endif() +endif() + +# Final status message +if (YAML_CPP_STATIC_FOUND) + message(STATUS "Using yaml-cpp STATIC library: ${YAML_CPP_LIBRARIES}") +elseif (YAML_CPP_SHARED_FOUND) + message(STATUS "Using yaml-cpp SHARED library: ${YAML_CPP_LIBRARIES}") +endif() + + +# Package dependencies - libpci3 is no longer a runtime dep (statically linked); +# yaml-cpp is conditionally required based on linking type +# RPM NUMA/yaml-cpp names differ by distro: +# RHEL/CentOS/Mariner: numactl-libs (runtime), yaml-cpp (dynamic) +# SLES 15/16: libnuma1 (runtime), libyaml-cpp0_6 (dynamic) +set(RVS_RPM_NUMA_DEPS "(numactl-libs or libnuma1)") +if (YAML_CPP_STATIC_FOUND) + # Static linking - no runtime yaml-cpp dependency needed + if (${RVS_OS_TYPE} STREQUAL "ubuntu") + set(CPACK_DEBIAN_PACKAGE_DEPENDS "${ROCM_DEB_DEPS}, libnuma1") + else () + # RPM distros - RHEL, CentOS, SLES, Mariner (manylinux-built RPM) + set(CPACK_RPM_PACKAGE_REQUIRES "${ROCM_RPM_DEPS}, ${RVS_RPM_NUMA_DEPS}") + endif() + message(STATUS "Package dependencies: yaml-cpp statically linked, no runtime dependency") +else() + # Dynamic linking - yaml-cpp runtime dependency required + if (${RVS_OS_TYPE} STREQUAL "ubuntu") + set(CPACK_DEBIAN_PACKAGE_DEPENDS "${ROCM_DEB_DEPS}, libyaml-cpp-dev, libnuma1") + else () + set(CPACK_RPM_PACKAGE_REQUIRES "${ROCM_RPM_DEPS}, (yaml-cpp or libyaml-cpp0_6), ${RVS_RPM_NUMA_DEPS}") + endif() + message(STATUS "Package dependencies: yaml-cpp dynamically linked, runtime dependency included") +endif() ################################################################################ ## GOOGLE TEST @@ -355,6 +526,7 @@ if(RVS_BUILD_TESTS) message(FATAL_ERROR "Build step for googletest failed: ${result}") endif() execute_process(COMMAND ${CMAKE_COMMAND} ${CMAKE_BINARY_DIR}/googletest-src -B${CMAKE_BINARY_DIR}/googletest-build + -DCMAKE_POSITION_INDEPENDENT_CODE=ON RESULT_VARIABLE result WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/googletest-src ) if(result) @@ -414,6 +586,90 @@ if (NOT DEFINED MXDATAGENERATOR_INC_DIR) endif () ################################################################################ +################################################################################ +# pciutils - build libpci.a from source to ensure ABI consistency across distros +# (pci_dev struct layout changed between libpci 3.7.0/Ubuntu-22.04 and +# 3.10.0/Ubuntu-24.04; pinning to v3.7.0 and static linking eliminates the +# runtime ABI mismatch and guarantees a clean build on Ubuntu 22.04 kernel headers) +configure_file(CMakePciutilsDownload.cmake pciutils-download/CMakeLists.txt) +execute_process(COMMAND ${CMAKE_COMMAND} -G "${CMAKE_GENERATOR}" . + RESULT_VARIABLE result + WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/pciutils-download ) +if(result) + message(FATAL_ERROR "CMake step for pciutils failed: ${result}") +endif() +execute_process(COMMAND ${CMAKE_COMMAND} --build . + RESULT_VARIABLE result + WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/pciutils-download ) +if(result) + message(FATAL_ERROR "Clone step for pciutils failed: ${result}") +endif() + +# pciutils uses a Makefile; SHARED=no produces libpci.a (static archive only). +# CFLAGS=-fPIC is required because libpci.a is linked into .so modules. +# ZLIB=no disables gzip ID-database support (gzopen/gzgets) - not needed by RVS. +# HWDB=no disables udev hardware-DB lookups (udev_new etc.) - not needed by RVS. +execute_process( + COMMAND make lib/libpci.a SHARED=no ZLIB=no HWDB=no CFLAGS=-fPIC + RESULT_VARIABLE result + WORKING_DIRECTORY ${CMAKE_BINARY_DIR}/pciutils-src +) +if(result) + message(FATAL_ERROR "Build step for pciutils (libpci.a) failed: ${result}") +endif() + +# Arrange headers into include/pci/ so #include resolves correctly +file(MAKE_DIRECTORY ${CMAKE_BINARY_DIR}/pciutils-install/include/pci) +file(GLOB PCIUTILS_HEADERS "${CMAKE_BINARY_DIR}/pciutils-src/lib/*.h") +file(COPY ${PCIUTILS_HEADERS} DESTINATION "${CMAKE_BINARY_DIR}/pciutils-install/include/pci/") + +# Imported target used by all modules via ${LIBPCI_TARGET} +add_library(pci-static STATIC IMPORTED GLOBAL) +set_target_properties(pci-static PROPERTIES + IMPORTED_LOCATION "${CMAKE_BINARY_DIR}/pciutils-src/lib/libpci.a" + INTERFACE_INCLUDE_DIRECTORIES "${CMAKE_BINARY_DIR}/pciutils-install/include" +) +set(LIBPCI_TARGET pci-static) +message(STATUS "pciutils: statically linking libpci.a built from v3.7.0") +################################################################################ + +################################################################################ +# TransferBench (vendored as a git submodule under external/TransferBench) +# +# Headers are consumed directly from the submodule by the PEBB and PBQT modules. +# The TransferBench CLI binary is built as an isolated sub-build and installed +# alongside rvs when BUILD_TRANSFERBENCH_CLI is ON (default). +################################################################################ +set(TRANSFERBENCH_SOURCE_DIR "${CMAKE_CURRENT_SOURCE_DIR}/external/TransferBench") + +if(NOT EXISTS "${TRANSFERBENCH_SOURCE_DIR}/CMakeLists.txt") + message(FATAL_ERROR + "TransferBench submodule not initialised. Run:\n" + " git submodule update --init --recursive\n" + "from the repository root and re-run cmake.") +endif() + +if(NOT DEFINED TRANSFERBENCH_INC_DIR) + set(TRANSFERBENCH_INC_DIR "${TRANSFERBENCH_SOURCE_DIR}/src/header") + message(STATUS "TRANSFERBENCH_INC_DIR ${TRANSFERBENCH_INC_DIR}") +endif() + +# TransferBench v1.69+: TransferBench.hpp includes IbvDynLoad.hpp from third-party/ibverbs. +# Omit when folder absent so older submodule pins remain buildable. +set(TRANSFERBENCH_IBVERBS_INC_DIR "") +if(EXISTS "${TRANSFERBENCH_SOURCE_DIR}/third-party/ibverbs") + set(TRANSFERBENCH_IBVERBS_INC_DIR "${TRANSFERBENCH_SOURCE_DIR}/third-party/ibverbs") + message(STATUS "TRANSFERBENCH_IBVERBS_INC_DIR ${TRANSFERBENCH_IBVERBS_INC_DIR}") +endif() + +option(BUILD_TRANSFERBENCH_CLI + "Build and install the TransferBench CLI alongside rvs" OFF) + +if(BUILD_TRANSFERBENCH_CLI) + include(${CMAKE_CURRENT_SOURCE_DIR}/CMakeTransferBenchCLI.cmake) +endif() +################################################################################ + ## rocBLAS if(RVS_ROCBLAS EQUAL 1) @@ -559,39 +815,36 @@ if(BUILD_ADDRESS_SANITIZER) set(ASAN_LD_FLAGS "-fuse-ld=lld") endif() -set(HCC_CXX_FLAGS "-fno-gpu-rdc --offload-arch=gfx90a --offload-arch=gfx1030 --offload-arch=gfx803 --offload-arch=gfx900 --offload-arch=gfx906 --offload-arch=gfx908 --offload-arch=gfx1101 --offload-arch=gfx942 --offload-arch=gfx1200 --offload-arch=gfx1201 --offload-arch=gfx1100 --offload-arch=gfx950") +set(HCC_CXX_FLAGS "-fno-gpu-rdc --offload-arch=gfx90a --offload-arch=gfx1030 --offload-arch=gfx803 --offload-arch=gfx900 --offload-arch=gfx906 --offload-arch=gfx908 --offload-arch=gfx1101 --offload-arch=gfx942 --offload-arch=gfx1200 --offload-arch=gfx1201 --offload-arch=gfx1100 --offload-arch=gfx950 --offload-arch=gfx1250") add_subdirectory(rvslib) add_subdirectory(rvs) -add_subdirectory(gm.so) add_subdirectory(gpup.so) add_subdirectory(gst.so) add_subdirectory(iet.so) add_subdirectory(pebb.so) add_subdirectory(peqt.so) -add_subdirectory(pesm.so) add_subdirectory(pbqt.so) add_subdirectory(rcqt.so) -add_subdirectory(smqt.so) add_subdirectory(mem.so) add_subdirectory(babel.so) -add_subdirectory(perf.so) add_subdirectory(tst.so) +add_subdirectory(pulse.so) if (RVS_BUILD_TESTS) add_subdirectory(testif.so) endif() -add_dependencies(pesm rvslib) if(RVS_ROCMSMI EQUAL 1) - add_dependencies(gm rvs_smi_target) add_dependencies(iet rvs_smi_target) + add_dependencies(pulse rvs_smi_target) endif() if(RVS_ROCBLAS EQUAL 1) add_dependencies(rvslib rvs_rblas_target) add_dependencies(gst rvs_rblas_target) add_dependencies(iet rvs_rblas_target) + add_dependencies(pulse rvs_rblas_target) endif() add_custom_target(rvs_bin_folder ALL @@ -658,7 +911,7 @@ install(DIRECTORY "${CMAKE_BINARY_DIR}/doc/man/man1/" ) install(DIRECTORY ${CMAKE_BINARY_DIR}/doc/userguide/html - DESTINATION ${CPACK_PACKAGING_INSTALL_PREFIX}/${CMAKE_INSTALL_DATADIR}/${CPACK_PACKAGE_NAME}/userguide + DESTINATION ${CPACK_PACKAGING_INSTALL_PREFIX}/${CMAKE_INSTALL_DATADIR}/${CMAKE_PROJECT_NAME}/userguide COMPONENT applications ) @@ -671,11 +924,11 @@ configure_file(${CMAKE_SOURCE_DIR}/rvs/conf/deviceid.sh.in ${CMAKE_SOURCE_DIR}/rvs/conf/deviceid.sh @ONLY) install(FILES ${CMAKE_SOURCE_DIR}/rvs/conf/deviceid.sh - DESTINATION ${CPACK_PACKAGING_INSTALL_PREFIX}/${CMAKE_INSTALL_DATADIR}/${CPACK_PACKAGE_NAME}/conf/ + DESTINATION ${CPACK_PACKAGING_INSTALL_PREFIX}/${CMAKE_INSTALL_DATADIR}/${CMAKE_PROJECT_NAME}/conf/ COMPONENT rvsmodule ) install(DIRECTORY ${CMAKE_SOURCE_DIR}/testscripts/ - DESTINATION ${CPACK_PACKAGING_INSTALL_PREFIX}/${CMAKE_INSTALL_DATADIR}/${CPACK_PACKAGE_NAME}/testscripts + DESTINATION ${CPACK_PACKAGING_INSTALL_PREFIX}/${CMAKE_INSTALL_DATADIR}/${CMAKE_PROJECT_NAME}/testscripts COMPONENT rvsmodule ) @@ -686,6 +939,7 @@ if(RVS_ROCMSMI EQUAL 1) install(FILES "${CMAKE_CURRENT_SOURCE_DIR}/rvsso.conf" DESTINATION /${CMAKE_INSTALL_SYSCONFDIR}/ld.so.conf.d COMPONENT rvsmodule) + set( COMP_TYPE "rvsmodule" ) endif() # if(RVS_ROCMSMI EQUAL 1) @@ -717,4 +971,14 @@ install(FILES set(CPACK_RPM_PACKAGE_AUTOREQ 0) +configure_pkg( ${CPACK_PACKAGE_NAME} ${COMP_TYPE} ${CPACK_PACKAGE_VERSION} ${PKG_MAINTAINER_NM} ${PKG_MAINTAINER_EMAIL} ) + +# Normalize RUNPATH on staged ELF before DEB/RPM/TGZ (strips build-machine SDK paths). +rvs_get_packaged_rpath_colon(RVS_PACKAGED_RPATH_COLON) +configure_file( + ${CMAKE_CURRENT_SOURCE_DIR}/cmake_modules/cpack-patch-rpath.cmake.in + ${CMAKE_BINARY_DIR}/cpack-patch-rpath.cmake + @ONLY) +set(CPACK_PRE_BUILD_SCRIPTS "${CMAKE_BINARY_DIR}/cpack-patch-rpath.cmake") + include (CPack) diff --git a/CMakeLists.txt.yaml b/CMakeLists.txt.yaml index d0d6575fb..4eb534249 100644 --- a/CMakeLists.txt.yaml +++ b/CMakeLists.txt.yaml @@ -23,7 +23,7 @@ ## ################################################################################ -cmake_minimum_required(VERSION 2.8.2) +cmake_minimum_required(VERSION 3.5.0) project(yaml-download NONE) diff --git a/CMakeMXDataGeneratorDownload.cmake b/CMakeMXDataGeneratorDownload.cmake index 0b9f9f8b5..8013ea648 100644 --- a/CMakeMXDataGeneratorDownload.cmake +++ b/CMakeMXDataGeneratorDownload.cmake @@ -23,7 +23,7 @@ ## ################################################################################ -cmake_minimum_required(VERSION 2.8.2) +cmake_minimum_required(VERSION 3.5.0) project(mxDataGenerator-download NONE) diff --git a/perf.so/tests.cmake b/CMakePciutilsDownload.cmake similarity index 67% rename from perf.so/tests.cmake rename to CMakePciutilsDownload.cmake index 84e6231fe..35898188d 100644 --- a/perf.so/tests.cmake +++ b/CMakePciutilsDownload.cmake @@ -1,6 +1,6 @@ ################################################################################ ## -## Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. +## Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. ## ## MIT LICENSE: ## Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -23,5 +23,19 @@ ## ################################################################################ +cmake_minimum_required(VERSION 3.5.0) -include(tests_conf_logging) +project(pciutils-download NONE) + +include(ExternalProject) +ExternalProject_Add(pciutils + GIT_REPOSITORY https://github.com/pciutils/pciutils.git + GIT_TAG "864aecdea9c7db626856d8d452f6c784316a878c" #Verion 3.7.0 + GIT_CONFIG "http.sslVerify=true" # Forces certificate validation + SOURCE_DIR "${CMAKE_BINARY_DIR}/pciutils-src" + BINARY_DIR "${CMAKE_BINARY_DIR}/pciutils-build" + CONFIGURE_COMMAND "" + BUILD_COMMAND "" + INSTALL_COMMAND "" + TEST_COMMAND "" +) diff --git a/CMakeRBLASDownload.cmake b/CMakeRBLASDownload.cmake index cfa2baf1f..305326897 100644 --- a/CMakeRBLASDownload.cmake +++ b/CMakeRBLASDownload.cmake @@ -23,7 +23,7 @@ ## ################################################################################ -cmake_minimum_required(VERSION 2.8.2) +cmake_minimum_required(VERSION 3.5.0) project(rvs_rblas-download NONE) diff --git a/CMakeRSMIDownload.cmake b/CMakeRSMIDownload.cmake index d105d6531..7f0460b34 100644 --- a/CMakeRSMIDownload.cmake +++ b/CMakeRSMIDownload.cmake @@ -23,7 +23,7 @@ ## ################################################################################ -cmake_minimum_required(VERSION 2.8.2) +cmake_minimum_required(VERSION 3.5.0) project(rvs_smi-download NONE) diff --git a/CMakeTransferBenchCLI.cmake b/CMakeTransferBenchCLI.cmake new file mode 100644 index 000000000..69e150fb9 --- /dev/null +++ b/CMakeTransferBenchCLI.cmake @@ -0,0 +1,153 @@ +################################################################################ +## +## Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +## MIT LICENSE +## +################################################################################ +# +# Build the TransferBench CLI binary from the external/TransferBench submodule +# as a sub-build of RVS, and install it alongside the rvs executable. +# +# TransferBench pins amdclang++ as the compiler before its project() call, so +# add_subdirectory would conflict with the parent compiler. ExternalProject_Add +# runs the child build in an isolated CMake invocation. +# +# The child build is BUILD-ONLY: we never invoke TransferBench's own install +# rules. TransferBench installs unconditionally via rocm_install() into an +# absolute prefix; letting it install into a build-tree directory caused that +# directory to leak verbatim into the packaged TGZ (e.g. +# __w/.../build/TransferBench-cli-install/bin/TransferBench). Instead we set +# INSTALL_COMMAND to a no-op and pick the binary straight out of the child +# build tree, then install it ourselves into the rvs package layout. +# +# TransferBench writes its CLI binary to the build-tree root via +# set(CMAKE_RUNTIME_OUTPUT_DIRECTORY .), so it lands at +# ${TRANSFERBENCH_BUILD_DIR}/TransferBench. +# +# We forward CMAKE_CXX_FLAGS so the child build inherits things like +# --gcc-toolchain=... that RVS's build script sets to make amdclang++ pick up +# a modern libstdc++ (required for C++20 in TransferBench). +# +# TRANSFERBENCH_GPU_TARGETS defaults to TransferBench build_packages_local.sh +# DEFAULT_GPU_TARGETS. Override with -DTRANSFERBENCH_GPU_TARGETS="..." or set +# GPU_TARGETS / TRANSFERBENCH_GPU_TARGETS in build_packages_local.sh. Set to +# empty to take TransferBench's own CMake default. GPU_TARGETS is written into +# the -C initial cache file (not -D on CMAKE_ARGS) because CMake list(APPEND) +# splits semicolon-separated strings. +# +# Feature flags match upstream TransferBench build_packages_local.sh (NIC/MPI/ +# DMA-BUF off, multi-arch not local-GPU-only). Upstream passes -DDISABLE_DMABUF +# but TransferBench CMake uses ENABLE_DMA_BUF instead (that upstream flag is a +# no-op). RPATH uses CMakeTransferBenchRPATH.cmake.in, not BUILD_RELOCATABLE_PACKAGE. +# +################################################################################ + +include(ExternalProject) + +if(NOT TRANSFERBENCH_SOURCE_DIR) + message(FATAL_ERROR "TRANSFERBENCH_SOURCE_DIR not set") +endif() + +set(TRANSFERBENCH_BUILD_DIR "${CMAKE_BINARY_DIR}/TransferBench-cli-build") + +# Source of truth: TransferBench build_packages_local.sh DEFAULT_GPU_TARGETS +set(TRANSFERBENCH_GPU_TARGETS + "gfx906;gfx908;gfx90a;gfx942;gfx950;gfx1030;gfx1100;gfx1101;gfx1102;gfx1150;gfx1151;gfx1200;gfx1201;gfx1250" + CACHE STRING "GPU targets to build TransferBench for") + +set(_tb_cmake_args + -DCMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE} + -DROCM_PATH=${ROCM_PATH} + -DHIP_PLATFORM=amd + -DCMAKE_VERBOSE_MAKEFILE=ON + -DBUILD_LOCAL_GPU_TARGET_ONLY=OFF + -DENABLE_NIC_EXEC=OFF + -DENABLE_MPI_COMM=OFF + -DENABLE_DMA_BUF=OFF +) + +# Forward parent CXX flags (carries --gcc-toolchain=... on RHEL/manylinux so +# amdclang++ picks up a modern libstdc++ with ). +if(CMAKE_CXX_FLAGS) + list(APPEND _tb_cmake_args "-DCMAKE_CXX_FLAGS=${CMAKE_CXX_FLAGS}") +endif() + +if(DEFINED ROCM_MAJOR_VERSION) + list(APPEND _tb_cmake_args -DROCM_MAJOR_VERSION=${ROCM_MAJOR_VERSION}) +endif() + +# Relocatable RUNPATH for the CLI (copied from the child build tree via install(PROGRAMS)). +# Parent ${CMAKE_INSTALL_RPATH} cannot be forwarded safely ($ORIGIN expands empty). +# TransferBench ignores CMAKE_INSTALL_RPATH; use -C cache with -Wl,-rpath only +# (see CMakeTransferBenchRPATH.cmake.in). GPU_TARGETS is also set in this file. +set(_tb_rpath_cache "${CMAKE_BINARY_DIR}/TransferBenchRPATH.cmake") +configure_file( + "${CMAKE_CURRENT_SOURCE_DIR}/CMakeTransferBenchRPATH.cmake.in" + "${_tb_rpath_cache}" + @ONLY +) +# On GitHub Actions skip build-tree RPATH additions (linker -rpath above is enough). +if("$ENV{GITHUB_ACTIONS}" STREQUAL "true") + file(APPEND "${_tb_rpath_cache}" + "set(CMAKE_SKIP_BUILD_RPATH TRUE CACHE BOOL \"\" FORCE)\n") +endif() +if(TRANSFERBENCH_GPU_TARGETS) + file(APPEND "${_tb_rpath_cache}" + "set(GPU_TARGETS [[${TRANSFERBENCH_GPU_TARGETS}]] CACHE STRING \"\" FORCE)\n") +endif() +list(APPEND _tb_cmake_args "-C${_tb_rpath_cache}") + +# Verbose child build: --verbose on cmake --build; stream to workflow log in CI. +set(_tb_build_command + ${CMAKE_COMMAND} --build ${TRANSFERBENCH_BUILD_DIR} --verbose) +if(CMAKE_BUILD_PARALLEL_LEVEL) + list(APPEND _tb_build_command --parallel ${CMAKE_BUILD_PARALLEL_LEVEL}) +endif() + +if("$ENV{GITHUB_ACTIONS}" STREQUAL "true") + set(_tb_log_configure OFF) + set(_tb_log_build OFF) +else() + set(_tb_log_configure ON) + set(_tb_log_build ON) +endif() + +message(STATUS "TransferBench CLI sub-build:") +message(STATUS " SOURCE_DIR=${TRANSFERBENCH_SOURCE_DIR}") +message(STATUS " BINARY_DIR=${TRANSFERBENCH_BUILD_DIR}") +message(STATUS " ROCM_PATH=${ROCM_PATH}") +message(STATUS " HIP_PLATFORM=amd") +message(STATUS " GPU_TARGETS=${TRANSFERBENCH_GPU_TARGETS}") +if(TRANSFERBENCH_GPU_TARGETS) + message(STATUS " GPU_TARGETS via: -C${_tb_rpath_cache}") +endif() +message(STATUS " CMAKE_BUILD_TYPE=${CMAKE_BUILD_TYPE}") +message(STATUS " BUILD_COMMAND: ${CMAKE_COMMAND} --build ${TRANSFERBENCH_BUILD_DIR} --verbose") +message(STATUS " LOG_CONFIGURE=${_tb_log_configure} LOG_BUILD=${_tb_log_build}") +foreach(_tb_arg IN LISTS _tb_cmake_args) + message(STATUS " CMAKE_ARGS: ${_tb_arg}") +endforeach() + +# Build only -- never run TransferBench's own install rules (see header). The +# binary is consumed directly from the build tree by the install(PROGRAMS) +# below, so there is no separate child install directory to leak into packages. +ExternalProject_Add(TransferBenchCLI + SOURCE_DIR "${TRANSFERBENCH_SOURCE_DIR}" + BINARY_DIR "${TRANSFERBENCH_BUILD_DIR}" + CMAKE_ARGS ${_tb_cmake_args} + BUILD_COMMAND ${_tb_build_command} + BUILD_ALWAYS OFF + INSTALL_COMMAND "" + TEST_COMMAND "" + LOG_CONFIGURE ${_tb_log_configure} + LOG_BUILD ${_tb_log_build} + LOG_OUTPUT_ON_FAILURE ON +) + +# Install the TransferBench CLI binary into the rvs package layout (same +# component as rvs itself, so it lands in the same DEB/RPM). Pulled straight +# from the child build tree -- TransferBench emits it at the build-dir root via +# set(CMAKE_RUNTIME_OUTPUT_DIRECTORY .). +install(PROGRAMS "${TRANSFERBENCH_BUILD_DIR}/TransferBench" + DESTINATION ${CPACK_PACKAGING_INSTALL_PREFIX}/${CMAKE_INSTALL_BINDIR} + COMPONENT rvsmodule) diff --git a/CMakeTransferBenchRPATH.cmake.in b/CMakeTransferBenchRPATH.cmake.in new file mode 100644 index 000000000..11792c3ee --- /dev/null +++ b/CMakeTransferBenchRPATH.cmake.in @@ -0,0 +1,9 @@ +# Initial cache fragment for the TransferBench ExternalProject sub-build. +# +# TransferBench links against ROCm install-tree libs and system libs (e.g. libnuma). +# Embed ROCm stack RUNPATH via linker flags only (TB does not honor CMAKE_INSTALL_RPATH). +set(CMAKE_SKIP_RPATH FALSE CACHE BOOL "" FORCE) +set(CMAKE_SKIP_INSTALL_RPATH TRUE CACHE BOOL "" FORCE) +set(CMAKE_INSTALL_RPATH_USE_LINK_PATH FALSE CACHE BOOL "" FORCE) +set(CMAKE_INSTALL_REMOVE_ENVIRONMENT_RPATH TRUE CACHE BOOL "" FORCE) +set(CMAKE_EXE_LINKER_FLAGS [[-Wl,--enable-new-dtags -Wl,-rpath,/opt/rocm/lib:/opt/rocm/core-@ROCM_MAJOR_VERSION@/lib]] CACHE STRING "" FORCE) diff --git a/DEBIAN/changelog.in b/DEBIAN/changelog.in new file mode 100644 index 000000000..057151379 --- /dev/null +++ b/DEBIAN/changelog.in @@ -0,0 +1,4 @@ +@DEB_PACKAGE_NAME@ (@DEB_PACKAGE_VERSION@) stable; urgency=low + + * ROCm Runtime software stack Base Package. + -- @DEB_MAINTAINER_NAME@ <@DEB_MAINTAINER_EMAIL@> @DEB_TIMESTAMP@ diff --git a/DEBIAN/copyright.in b/DEBIAN/copyright.in new file mode 100644 index 000000000..2dfae2cc1 --- /dev/null +++ b/DEBIAN/copyright.in @@ -0,0 +1,25 @@ +Format: https://www.debian.org/doc/packaging-manuals/copyright-format/1.0/ +Upstream-Name: @DEB_PACKAGE_NAME@ +Upstream-Contact: @DEB_MAINTAINER_NAME@ <@DEB_MAINTAINER_EMAIL@> +Source: https://github.com/ROCm/@DEB_PACKAGE_NAME@ +Files: * +License: @DEB_LICENSE@ +Copyright: @DEB_COPYRIGHT_YEAR@ Advanced Micro Devices, Inc. All rights Reserved. + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/Doxyfile b/Doxyfile index 92c5341b1..6eb7ec13f 100644 --- a/Doxyfile +++ b/Doxyfile @@ -781,7 +781,7 @@ WARN_LOGFILE = # spaces. See also FILE_PATTERNS and EXTENSION_MAPPING # Note: If this tag is empty the current directory is searched. -INPUT = rvs src include gpup.so gm.so pesm.so rcqt.so peqt.so smqt.so \ +INPUT = rvs src include gpup.so rcqt.so peqt.so \ pbqt.so pebb.so gst.so iet.so diff --git a/FEATURES.md b/FEATURES.md index adae880de..b9058cf02 100644 --- a/FEATURES.md +++ b/FEATURES.md @@ -4,15 +4,6 @@ ## GPU Properties – GPUP The GPU Properties module queries the configuration of a target device and returns the device’s static characteristics. These static values can be used to debug issues such as device support, performance and firmware problems. -## GPU Monitor – GM module [deprecated] -The GPU monitor tool is capable of running on one, some or all of the GPU(s) installed and will report various information at regular intervals. The module can be configured to halt another RVS modules execution if one of the quantities exceeds a specified boundary value. - -## PCI Express State Monitor – PESM module [deprecated] -The PCIe State Monitor tool is used to actively monitor the PCIe interconnect between the host platform and the GPU. The module will register a “listener” on a target GPU’s PCIe interconnect, and log a message whenever it detects a state change. The PESM will be able to detect the following state changes: - -1. PCIe link speed changes -2. GPU power state changes - ## ROCm Configuration Qualification Tool - RCQT module The ROCm Configuration Qualification Tool ensures the platform is capable of running ROCm applications and is configured correctly. It checks the installed versions of the ROCm components and the platform configuration of the system. This includes checking the dependencies corresponding to the ROCm meta-packages are installed correctly. @@ -24,9 +15,6 @@ The PCIe Qualification Tool is used to qualify the PCIe bus on which the GPU is 3. PCIe link speed 4. PCIe link width -## SBIOS Mapping Qualification Tool – SMQT module [deprecated] -The GPU SBIOS mapping qualification tool is designed to verify that a platform’s SBIOS has satisfied the BAR mapping requirements for VDI and Radeon Instinct products for ROCm support. - ## P2P Benchmark and Qualification Tool – PBQT module The P2P Benchmark and Qualification Tool is designed to provide the list of all GPUs that support P2P and characterize the P2P links between peers. In addition to testing P2P compatibility, this test will perform a peer-to-peer throughput test between all P2P pairs for performance evaluation. The P2P Benchmark and Qualification Tool will allow users to pick a collection of two or more GPUs to run the test. The user will also be able to select whether or not they want to run the throughput test on each of the pairs. @@ -59,3 +47,7 @@ The Babel module executes BabelStream (synthetic GPU benchmark based on the orig ## Thermal Stress Test - TST module The Thermal Stress Test (TST) measures/monitors the GPU edge and junction temperatures under various stressful workloads. Also checks whether GPU junction temperature reaches target and trottle temperatures. +## Pulse Stress Test- Pulse module +The Pulse Stress Test creates power fluctuations by alternating between high-compute (GEMM) and idle phases at a configurable rate. Two-level barrier synchronizes all GPUs so they spike current simultaneously, maximizing stress on the PSU. + +**PS: Beta version — not intended for production use. Pass/fail criteria are still being tuned.** diff --git a/README.md b/README.md index 9732e084c..bb0347513 100644 --- a/README.md +++ b/README.md @@ -1,7 +1,7 @@ # ROCmValidationSuite -The ROCm Validation Suite (RVS) is a system validation and diagnostics tool for monitoring, stress testing, detecting and troubleshooting issues that affects the functionality and performance of AMD GPU(s) operating in a high-performance/AI/ML computing environment. RVS is enabled using the ROCm software stack on a compatible software and hardware platform. +The ROCm Validation Suite (RVS) is a system validation and diagnostics tool for monitoring, stress testing, detecting and troubleshooting issues that affect the functionality and performance of AMD GPU(s) operating in a high-performance/AI/ML computing environment. RVS is enabled using the ROCm software stack on a compatible software and hardware platform. -RVS is a collection of tests, benchmarks and qualification tools each targeting a specific sub-system of the ROCm platform. All of the tools are implemented in software and share a common command line interface. Each set of tests are implemented in a “module” which is a library encapsulating the functionality specific to the tool. The CLI can specify the directory containing modules to use when searching for libraries to load. Each module may have a set of options that it defines and a configuration file that supports its execution. +RVS is a collection of tests, benchmarks and qualification tools each targeting a specific sub-system of the ROCm platform. All of the tools are implemented in software and share a common command-line interface. Each set of tests is implemented in a “module” which is a library encapsulating the functionality specific to the tool. The CLI can specify the directory containing modules to use when searching for libraries to load. Each module may have a set of options that it defines and a configuration file that supports its execution. For different RVS modules and their description, refer to [the documentation on features](./FEATURES.md). @@ -13,37 +13,45 @@ Please do this before compilation/installing compiled package. Ubuntu : ``` -sudo apt-get -y update && sudo apt-get install -y libpci3 libpci-dev doxygen unzip cmake git libyaml-cpp-dev +sudo apt-get -y update && sudo apt-get install -y libpci3 libpci-dev doxygen unzip cmake git libyaml-cpp-dev libnuma-dev +``` + +**Note:** RVS requires CMake >= 3.25. Ubuntu 22.04 and earlier ship with an older version (3.22). If you are on Ubuntu 22.04 or earlier, install a newer CMake from Kitware's official APT repository before proceeding: + +``` +wget -O - https://apt.kitware.com/keys/kitware-archive-latest.asc | sudo apt-key add - +sudo apt-add-repository 'deb https://apt.kitware.com/ubuntu/ jammy main' +sudo apt-get update && sudo apt-get install cmake ``` CentOS : ``` -sudo yum install -y cmake3 doxygen pciutils-devel rpm rpm-build git gcc-c++ yaml-cpp-devel +sudo yum install -y cmake3 doxygen pciutils-devel rpm rpm-build git gcc-c++ yaml-cpp-devel yaml-cpp-static numactl-devel ``` RHEL : ``` -sudo yum install -y cmake3 doxygen rpm rpm-build git gcc-c++ yaml-cpp-devel pciutils-devel +sudo yum install -y cmake3 doxygen rpm rpm-build git gcc-c++ yaml-cpp-devel yaml-cpp-static pciutils-devel numactl-devel ``` SLES : ``` -sudo zypper install -y cmake doxygen pciutils-devel libpci3 rpm git rpm-build gcc-c++ yaml-cpp-devel +sudo zypper install -y cmake doxygen pciutils-devel libpci3 rpm git rpm-build gcc-c++ yaml-cpp-devel libnuma-devel ``` -## Install ROCm stack, rocblas and rocm-smi-lib +**NUMA libraries:** build headers are `libnuma-dev` (Ubuntu), `numactl-devel` (RHEL/CentOS), and `libnuma-devel` (SLES 15/16). Packaged DEB/RPM runtime deps use `libnuma1` on Ubuntu and SLES, and `numactl-libs` on RHEL/CentOS; RPM metadata uses `(numactl-libs or libnuma1)` so one manylinux-built RPM works on both families. + +## Install ROCm stack, rocBLAS, and SMI library + Install ROCm stack for Ubuntu/CentOS/SLES/RHEL. Refer to [ROCm installation guide](https://rocmdocs.amd.com/en/latest/Installation_Guide/Installation-Guide.html) for more details. -**Note:** - -rocm_smi64 package has been renamed to rocm-smi-lib64 from >= ROCm3.0. If you are using ROCm release < 3.0 , install the package as "rocm_smi64". -rocm-smi-lib64 package has been renamed to rocm-smi-lib from >= ROCm4.1. +Install rocBLAS and the SMI library. For ROCm 6.4 and earlier, install `rocm-smi-lib`. For ROCm 7.0 and later, install `amd-smi-lib`. -Install rocBLAS and rocm-smi-lib : +**ROCm 6.4 and earlier:** Ubuntu : @@ -63,8 +71,28 @@ SUSE : sudo zypper install rocblas rocm-smi-lib ``` +**ROCm 7.0 and later:** + +Ubuntu : + +``` +sudo apt-get install rocblas amd-smi-lib +``` + +CentOS & RHEL : + +``` +sudo yum install --nogpgcheck rocblas amd-smi-lib +``` + +SUSE : + +``` +sudo zypper install rocblas amd-smi-lib +``` + **Note:** -If rocm-smi-lib is already installed but /opt/rocm/lib/librocm_smi64.so doesn't exist. Do below: +If `rocm-smi-lib` is already installed but `/opt/rocm/lib/librocm_smi64.so` does not exist, reinstall it as follows: Ubuntu : @@ -75,13 +103,13 @@ sudo dpkg -r rocm-smi-lib && sudo apt install rocm-smi-lib CentOS & RHEL : ``` -sudo rpm -e rocm-smi-lib && sudo yum install rocm-smi-lib +sudo rpm -e rocm-smi-lib && sudo yum install rocm-smi-lib ``` SUSE : ``` -sudo rpm -e rocm-smi-lib && sudo zypper install rocm-smi-lib +sudo rpm -e rocm-smi-lib && sudo zypper install rocm-smi-lib ``` ## Building from Source @@ -91,10 +119,27 @@ This section explains how to get and compile current development stream of RVS. ### Clone repository ``` -git clone https://github.com/ROCm/ROCmValidationSuite.git +git clone --recurse-submodules https://github.com/ROCm/ROCmValidationSuite.git ``` -**Note:** +If the repository was already cloned without `--recurse-submodules`, initialise the TransferBench submodule from inside the clone: + +``` +git submodule update --init --recursive +``` + +### Bundled TransferBench + +[TransferBench](https://github.com/ROCm/TransferBench) is vendored under `external/TransferBench` as a git submodule. The standalone CLI is **off by default** in [`build_packages_local.sh`](build_packages_local.sh) (`BUILD_TRANSFERBENCH_CLI=OFF`); **CI builds enable it by default** so DEB/RPM/TGZ ship `TransferBench` alongside `rvs` unless disabled via repository variable or workflow input. + +The bundled CLI is provided **for compatibility** with existing workflows that invoke `TransferBench` directly. For new work, prefer either: + +- **RVS**, which exposes TransferBench functionality through the `pebb` and `pbqt` modules with config-driven test definitions, or +- The **TransferBench API** (headers under `external/TransferBench/src/header`) for programmatic use. + +Enable the CLI locally with `BUILD_TRANSFERBENCH_CLI=ON ./build_packages_local.sh` or `-DBUILD_TRANSFERBENCH_CLI=ON` when invoking cmake directly. + +**Note:** The above command clones the master branch. If you're using a specific ROCm release, it's recommended to use the corresponding RVS version from the same release branch to ensure compatibility. If ROCm 6.4 is installed, clone the RVS repository from the 6.4 release branch by running: @@ -103,7 +148,7 @@ If ROCm 6.4 is installed, clone the RVS repository from the 6.4 release branch b git clone https://github.com/ROCm/ROCmValidationSuite.git -b release/rocm-rel-6.4 ``` -### Configure: +### Configure ``` cd ROCmValidationSuite @@ -125,16 +170,16 @@ Option 2: Using the symbolic link cmake -B ./build -DROCM_PATH=/opt/rocm -DCMAKE_INSTALL_PREFIX=/opt/rocm -DCPACK_PACKAGING_INSTALL_PREFIX=/opt/rocm ``` -### Build binary: +### Build binary ``` -make -C ./build +make -C ./build -j $(nproc) ``` **Note:** -Use the -j option with the build command to enable parallel compilation, which can significantly speed up the build process. +`$(nproc)` automatically uses all available CPU cores for parallel compilation, which can significantly speed up the build process. You can replace it with a specific number (e.g., `-j 8`) to limit core usage. -### Build package: +### Build package ``` cd ./build @@ -144,7 +189,7 @@ make package **Note:** Based on your OS, only DEB or RPM package will be built. You may ignore an error for the unrelated configuration -### Install built package: +### Install built package Ubuntu : @@ -158,11 +203,13 @@ CentOS & RHEL & SUSE : sudo rpm -i --replacefiles --nodeps rocm-validation-suite*.rpm ``` +If the package manager reports missing NUMA libraries, install the runtime package for your OS before retrying: `libnuma1` on Ubuntu/SLES, or `numactl-libs` on RHEL/CentOS (`zypper install libnuma1` / `dnf install numactl-libs`). + **Note:** -RVS is getting packaged as part of ROCm release starting from 3.0. You can install pre-compiled package as below. +RVS is getting packaged as part of ROCm release starting from 3.0. You can install the pre-compiled package as below. Please make sure Prerequisites, ROCm stack, rocblas and rocm-smi-lib64 are already installed -### Install package packaged with ROCm release: +### Install package packaged with ROCm release Ubuntu : @@ -182,6 +229,56 @@ SUSE : sudo zypper install rocm-validation-suite ``` +## Install RVS from Tarball - for TheRock based ROCm installation + +Follow these steps to install RVS from a prebuilt tarball on top of an existing ROCm Core SDK installation using the TheRock build system. + +### Download the RVS tarball + +``` +wget https://repo.amd.com/rocm/rvs/tarball/amdrocm7-rvs-1.4.21-288-Linux.tar.gz +``` + +### Extract the tarball + +Set `ROCM_PATH` to your ROCm Core SDK location. For a package manager installation, the default is `/opt/rocm`: + +``` +export ROCM_PATH=/opt/rocm +sudo mkdir -p $ROCM_PATH/extras-7 +sudo tar -xzf amdrocm7-rvs-1.4.21-288-Linux.tar.gz -C $ROCM_PATH/extras-7 +``` + +### Set up your environment + +**User setup** + +Add the following to `~/.bashrc`, then run `source ~/.bashrc`: + +``` +export ROCM_PATH=/opt/rocm +export PATH=$ROCM_PATH/extras-7/bin:$ROCM_PATH/bin:$PATH +export LD_LIBRARY_PATH=$ROCM_PATH/extras-7/lib:$ROCM_PATH/lib:$ROCM_PATH/lib/llvm/lib:$LD_LIBRARY_PATH +``` + +**System-wide setup:** + +``` +sudo tee /etc/profile.d/set-rocm-env.sh << EOF +export ROCM_PATH=/opt/rocm +export PATH=\$ROCM_PATH/extras-7/bin:\$ROCM_PATH/bin:\$PATH +export LD_LIBRARY_PATH=\$ROCM_PATH/extras-7/lib:\$ROCM_PATH/lib:\$ROCM_PATH/lib/llvm/lib:\$LD_LIBRARY_PATH +EOF +sudo chmod +x /etc/profile.d/set-rocm-env.sh +source /etc/profile.d/set-rocm-env.sh +``` + +### Verify the installation + +``` +rvs -h +``` + ## Running RVS ### Run version built from source code @@ -193,9 +290,11 @@ cd /build/bin Command examples: ``` -./rvs --help ; Lists all options to run RVS test suite -./rvs -g ; Lists supported GPUs available in the machine -./rvs -c conf/gst_single.conf ; Run GST module default test configuration +./rvs --help # Lists all options to run RVS test suite +./rvs -g # Lists supported GPUs available in the machine +./rvs -c conf/gst_single.conf # Run GST module default test configuration +./rvs -m gst # Run GST module using platform-detected config +./rvs -r 3 # Run RVS level 3 (range: 1–5, 5 = highest) tests for the platform-detected ``` ### Run version pre-compiled and packaged with ROCm release @@ -207,22 +306,24 @@ cd /opt/rocm/bin Command examples: ``` -./rvs --help ; Lists all options to run RVS test suite -./rvs -g ; Lists supported GPUs available in the machine -./rvs -c ../share/rocm-validation-suite/conf/gst_single.conf ; Run GST default test configuration +./rvs --help # Lists all options to run RVS test suite +./rvs -g # Lists supported GPUs available in the machine +./rvs -c ../share/rocm-validation-suite/conf/gst_single.conf # Run GST default test configuration +./rvs -m gst # Run GST module using platform-detected config +./rvs -r 3 # Run RVS level 3 (range: 1–5, 5 = highest) tests for the platform-detected ``` To run GPU specific test configuration, use configuration files from GPU folders in "/opt/rocm/share/rocm-validation-suite/conf" ``` -./rvs -c ../share/rocm-validation-suite/conf/MI300X/gst_single.conf ; Run MI300X specific GST test configuration -./rvs -c ../share/rocm-validation-suite/conf/nv32/gst_single.conf ; Run Navi 32 specific GST test configuration +./rvs -c ../share/rocm-validation-suite/conf/MI300X/gst_single.conf # Run MI300X specific GST test configuration +./rvs -c ../share/rocm-validation-suite/conf/nv32/gst_single.conf # Run Navi 32 specific GST test configuration ``` -**Note:** +**Note:** If present, always use GPU specific configurations instead of default test configurations. ## Reporting -Test results, errors and verbose logs are printed as terminal output. To enable json logging use "-j" command line option. -The json output file path can be specified after "-j" option. If not specified, a file is created in /var/tmp folder and the name of the file will be printed to stdout. The json file schemas are documented at [schemas](./docs/schemas) +Test results, errors and verbose logs are printed as terminal output. To enable JSON logging, use the `-j` command-line option. +The JSON output file path can be specified after the `-j` option. If not specified, a file is created in the `/var/tmp` folder and the name of the file will be printed to stdout. The JSON file schemas are documented at [schemas](./docs/schemas) diff --git a/babel.so/CMakeLists.txt b/babel.so/CMakeLists.txt index 4f7289744..08c3355c7 100644 --- a/babel.so/CMakeLists.txt +++ b/babel.so/CMakeLists.txt @@ -60,7 +60,7 @@ set(HIP_HCC_BUILD_FLAGS "${HIP_HCC_BUILD_FLAGS} -DHIP_VERSION_MAJOR=${HIP_VERSIO set(HIP_HCC_BUILD_FLAGS) set(HIP_HCC_BUILD_FLAGS "${HIP_HCC_BUILD_FLAGS} -fPIC ${HCC_CXX_FLAGS} -I${HSA_INC_DIR} ${ASAN_CXX_FLAGS}") -set(HIP_STREAM_BUILD_FLAGS "-DNONTEMPORAL=1 -O3 -std=c++17") +set(HIP_STREAM_BUILD_FLAGS "-O3 -std=c++17") # Set compiler and compiler flags set(CMAKE_CXX_COMPILER "${HIPCC_PATH}/bin/hipcc") @@ -145,7 +145,7 @@ include_directories(./ ../ ${ROCR_INC_DIR} ${HIP_INC_DIR}) # Add directories to look for library files to link link_directories(${RVS_LIB_DIR} ${ROCR_LIB_DIR} ${ROCBLAS_LIB_DIR} ${ASAN_LIB_PATH} ${AMD_SMI_LIB_DIR} ${HIPRAND_LIB_DIR} ${ROCRAND_LIB_DIR}) ## additional libraries -set (PROJECT_LINK_LIBS rvslib libpthread.so libpci.so libm.so) +set (PROJECT_LINK_LIBS rvslib libpthread.so libm.so) ## define source files set(SOURCES src/rvs_module.cpp src/action.cpp src/rvs_stress.cpp src/rvs_stream.cpp src/rvs_memworker.cpp) diff --git a/babel.so/include/HIPStream.h b/babel.so/include/HIPStream.h index 8d254d09e..1d85622b4 100644 --- a/babel.so/include/HIPStream.h +++ b/babel.so/include/HIPStream.h @@ -11,6 +11,8 @@ #include #include #include +#include +#include #include "Stream.h" #include "hip/hip_runtime.h" @@ -47,9 +49,17 @@ class HIPStream : public Stream T *d_b; T *d_c; + int nt_mode = 1; // NT_ALL + + // sustained mode: skip per-iteration sync, single sync after all iterations. + bool sustained_mode = false; + public: + void set_sustained_mode(bool mode) { sustained_mode = mode; } + void sustained_sync(); HIPStream(const unsigned int, const bool, const int, - const unsigned int, const unsigned int, const unsigned int); + const unsigned int, const unsigned int, const unsigned int, + const std::string& nontemporal = "all"); ~HIPStream(); virtual float read() override; @@ -61,6 +71,8 @@ class HIPStream : public Stream virtual T dot() override; virtual void init_arrays(T initA, T initB, T initC) override; + virtual void init_arrays_normdist(T mean, T stddev, bool gpu_init, + std::vector& a, std::vector& b, std::vector& c); virtual void read_arrays(std::vector& a, std::vector& b, std::vector& c) override; }; diff --git a/babel.so/include/action.h b/babel.so/include/action.h index fae960944..ecbc7a389 100644 --- a/babel.so/include/action.h +++ b/babel.so/include/action.h @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2025 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -25,13 +25,6 @@ #ifndef MEM_SO_INCLUDE_ACTION_H_ #define MEM_SO_INCLUDE_ACTION_H_ -#ifdef __cplusplus -extern "C" { -#endif -#include -#ifdef __cplusplus -} -#endif #include #include @@ -52,22 +45,34 @@ using std::map; #define RVS_CONF_TEST_TYPE "test_type" #define RVS_CONF_MEM_MIBIBYTE "mibibytes" #define RVS_CONF_OP_CSV "o/p_csv" -#define RVS_CONF_RWTEST "rwtest" -#define RVS_CONF_SUBTEST "subtest" #define RVS_CONF_DWORDS_PER_LANE "dwords_per_lane" #define RVS_CONF_CHUNKS_PER_BLOCK "chunks_per_block" #define RVS_CONF_TB_SIZE "tb_size" +#define RVS_CONF_READ "read" +#define RVS_CONF_WRITE "write" +#define RVS_CONF_COPY "copy" +#define RVS_CONF_ADD "add" +#define RVS_CONF_MUL "mul" +#define RVS_CONF_DOT "dot" +#define RVS_CONF_TRIAD "triad" +#define RVS_CONF_DATA_INIT "data_init" +#define RVS_CONF_NONTEMPORAL "nontemporal" +#define RVS_CONF_DURATION "duration" +#define RVS_CONF_SUSTAINED "sustained" #define MEM_DEFAULT_ARRAY_SIZE 33554432 // 32 MB #define MEM_DEFAULT_NUM_ITER 100 +#define MEM_DEFAULT_DURATION 0 // 0 means use num_iter instead #define MEM_DEFAULT_TEST_TYPE 1 #define MEM_DEFAULT_MEM_MIBIBYTE false #define MEM_DEFAULT_OP_CSV false -#define MEM_DEFAULT_RWTEST 0 -#define MEM_DEFAULT_SUBTEST 5 #define MEM_DEFAULT_DWORDS_PER_LANE 4 #define MEM_DEFAULT_CHUNKS_PER_BLOCK 2 #define MEM_DEFAULT_TB_SIZE 1024 +#define MEM_DEFAULT_TEST_ENABLE false +#define MEM_DEFAULT_DATA_INIT "default" +#define MEM_DEFAULT_NONTEMPORAL "all" +#define MEM_DEFAULT_SUSTAINED false #define MEM_NO_COMPATIBLE_GPUS "No AMD compatible GPU found!" #define FLOATING_POINT_REGEX "^[0-9]*\\.?[0-9]+$" @@ -98,15 +103,27 @@ class mem_action: public rvs::actionbase { bool mibibytes; //! output in csv bool output_csv; + //! read test enable/disable + bool read; + //! write test enable/disable + bool write; + //! copy test enable/disable + bool copy; + //! add test enable/disable + bool add; + //! mul test enable/disable + bool mul; + //! dot test enable/disable + bool dot; + //! triad test enable/disable + bool triad; //! test type int test_type; - //read-write test selection - int rwtest; - //subtest selection - int subtest; //! number of iterations uint64_t num_iterations; - //! number of iterations + //! test duration in milliseconds (0 = use num_iterations) + uint64_t duration; + //! array size uint64_t array_size; //! number of dwords per lane uint16_t dwords_per_lane; @@ -114,6 +131,12 @@ class mem_action: public rvs::actionbase { uint16_t chunks_per_block; //! thread block size uint16_t tb_size; + //! data initialization mode ("gpu_norm_dist", "cpu_norm_dist", "zero_init" or "default") + std::string data_init; + //! non-temporal access mode ("none", "all", "read" or "write") + std::string nontemporal; + //! sustained mode: back-to-back launches, single sync, average BW only + bool sustained; // configuration properties getters bool get_all_mem_config_keys(void); diff --git a/babel.so/include/rvs_memworker.h b/babel.so/include/rvs_memworker.h index 144da54a2..29dbba2a2 100644 --- a/babel.so/include/rvs_memworker.h +++ b/babel.so/include/rvs_memworker.h @@ -103,6 +103,17 @@ #define TRAID_FLOAT 3 #define TRIAD_DOUBLE 4 +/* Babel subtest enable/disable */ +typedef struct { + bool read; + bool write; + bool copy; + bool add; + bool mul; + bool dot; + bool triad; +} subtest; + /** * @class MEMWorker * @ingroup MEM @@ -162,6 +173,13 @@ class MemWorker : public rvs::ThreadBase { //! returns the number of iterations uint64_t get_num_iterations(void) { return num_iterations; } + //! sets the test duration in milliseconds (0 = use num_iterations) + void set_duration(uint64_t _duration) { + duration = _duration; + } + //! returns the test duration in milliseconds + uint64_t get_duration(void) { return duration; } + //! sets the array size void set_array_size(uint64_t _array_size) { array_size = _array_size; @@ -176,19 +194,40 @@ class MemWorker : public rvs::ThreadBase { //! returns the test type int get_test_type(void) { return test_type; } - //! sets rw test type - void set_rwtest_type(int _test_type) { - rwtest = _test_type; + //! sets read test enable/disable + void set_read(bool _read) { + read = _read; + } + + //! sets write test enable/disable + void set_write(bool _write) { + write = _write; + } + + //! sets copy test enable/disable + void set_copy(bool _copy) { + copy = _copy; + } + + //! sets add test enable/disable + void set_add(bool _add) { + add = _add; } - //! returns rw test type - int get_rwtest_type(void) { return rwtest; } - //! sets the sub test type - void set_subtest_type(int _test_type) { - subtest = _test_type; + //! sets mul test enable/disable + void set_mul(bool _mul) { + mul = _mul; + } + + //! sets dot test enable/disable + void set_dot(bool _dot) { + dot = _dot; + } + + //! sets triad test enable/disable + void set_triad(bool _triad) { + triad = _triad; } - //! returns the sub test type - int get_subtest_type(void) { return subtest; } //! sets the mibi bytes void set_mibibytes(bool _mibibytes) { @@ -223,9 +262,30 @@ class MemWorker : public rvs::ThreadBase { tb_size = _tb_size; } + //! sets data initialization mode + void set_data_init(const std::string& _data_init) { + data_init = _data_init; + } + //! returns data initialization mode + const std::string& get_data_init(void) { return data_init; } + + //! sets non-temporal access mode + void set_nontemporal(const std::string& _nontemporal) { + nontemporal = _nontemporal; + } + //! returns non-temporal access mode + const std::string& get_nontemporal(void) { return nontemporal; } + + //! sets sustained mode + void set_sustained(bool _sustained) { sustained = _sustained; } + //! returns sustained mode + bool get_sustained(void) { return sustained; } + static void set_use_json(bool _bjson) { bjson = _bjson; } //! returns the JSON flag static bool get_use_json(void) { return bjson; } + //! get worker job result + bool get_result(void) { return result; } protected: bool do_mem_stress_test(int *error, std::string *err_description); @@ -247,6 +307,8 @@ class MemWorker : public rvs::ThreadBase { uint64_t run_duration_ms; //! Number of iterations uint64_t num_iterations; + //! Test duration in milliseconds (0 = use num_iterations) + uint64_t duration; //! output as csv bool output_csv; //! Mibibytes @@ -255,21 +317,40 @@ class MemWorker : public rvs::ThreadBase { uint64_t array_size; //! Test type int test_type; - //! Read-Write Test type - int rwtest; - //! Sub Test type - int subtest; //! number of dwords per lane uint16_t dwords_per_lane; //! number of chunks per block uint16_t chunks_per_block; //! thread block size uint16_t tb_size; + //! data initialization mode + std::string data_init; + //! non-temporal access mode + std::string nontemporal; + //! sustained mode: back-to-back launches, single sync, average BW only + bool sustained; //! TRUE if JSON output is required static bool bjson; //! synchronization mutex std::mutex wrkrmutex; + //! Worker job result + bool result; + + //! read test enable/disable + bool read; + //! write test enable/disable + bool write; + //! copy test enable/disable + bool copy; + //! add test enable/disable + bool add; + //! mul test enable/disable + bool mul; + //! dot test enable/disable + bool dot; + //! triad test enable/disable + bool triad; }; #endif // MEM_SO_INCLUDE_MEM_WORKER_H_ diff --git a/babel.so/src/action.cpp b/babel.so/src/action.cpp index f7069e502..b80083af9 100644 --- a/babel.so/src/action.cpp +++ b/babel.so/src/action.cpp @@ -68,16 +68,16 @@ mem_action::~mem_action() { * @return true if no error occured, false otherwise */ bool mem_action::do_mem_stress_test(map mem_gpus_device_index) { - size_t k = 0; + + uint64_t k = 0; string msg; + vector workers(mem_gpus_device_index.size()); for (;;) { - unsigned int i = 0; if (property_wait != 0) // delay mem execution sleep(property_wait); - vector workers(mem_gpus_device_index.size()); - + size_t i = 0; map::iterator it; // all worker instances have the same json settings @@ -101,11 +101,20 @@ bool mem_action::do_mem_stress_test(map mem_gpus_device_index) { workers[i].set_mibibytes(mibibytes); workers[i].set_output_csv(output_csv); workers[i].set_num_iterations(num_iterations); - workers[i].set_rwtest_type(rwtest); - workers[i].set_subtest_type(subtest); + workers[i].set_duration(duration); + workers[i].set_read(read); + workers[i].set_write(write); + workers[i].set_copy(copy); + workers[i].set_add(add); + workers[i].set_mul(mul); + workers[i].set_dot(dot); + workers[i].set_triad(triad); workers[i].set_dwords_per_lane(dwords_per_lane); workers[i].set_chunks_per_block(chunks_per_block); workers[i].set_tb_size(tb_size); + workers[i].set_data_init(data_init); + workers[i].set_nontemporal(nontemporal); + workers[i].set_sustained(sustained); i++; } @@ -139,7 +148,18 @@ bool mem_action::do_mem_stress_test(map mem_gpus_device_index) { } } - return rvs::lp::Stopping() ? false : true; + if (rvs::lp::Stopping()) { + return false; + } + else { + for (size_t i = 0; i < mem_gpus_device_index.size(); i++) { + if(false == workers[i].get_result()) { + return false; + } + } + } + + return true; } /** @@ -173,19 +193,58 @@ bool mem_action::get_all_mem_config_keys(void) { rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); bsts = false; } + if (property_get(RVS_CONF_READ, + &read, MEM_DEFAULT_TEST_ENABLE)) { + msg = "invalid '" + + std::string(RVS_CONF_READ) + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + if (property_get(RVS_CONF_WRITE, + &write, MEM_DEFAULT_TEST_ENABLE)) { + msg = "invalid '" + + std::string(RVS_CONF_WRITE) + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + if (property_get(RVS_CONF_COPY, + ©, MEM_DEFAULT_TEST_ENABLE)) { + msg = "invalid '" + + std::string(RVS_CONF_COPY) + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + if (property_get(RVS_CONF_ADD, + &add, MEM_DEFAULT_TEST_ENABLE)) { + msg = "invalid '" + + std::string(RVS_CONF_ADD) + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + if (property_get(RVS_CONF_MUL, + &mul, MEM_DEFAULT_TEST_ENABLE)) { + msg = "invalid '" + + std::string(RVS_CONF_MUL) + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } - if (property_get_int(RVS_CONF_RWTEST, - &rwtest, MEM_DEFAULT_RWTEST)) { + if (property_get(RVS_CONF_DOT, + &dot, MEM_DEFAULT_TEST_ENABLE)) { msg = "invalid '" + - std::string(RVS_CONF_RWTEST) + "' key value"; + std::string(RVS_CONF_DOT) + "' key value"; rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); bsts = false; } - if (property_get_int(RVS_CONF_SUBTEST, - &subtest, MEM_DEFAULT_SUBTEST)) { + if (property_get(RVS_CONF_TRIAD, + &triad, MEM_DEFAULT_TEST_ENABLE)) { msg = "invalid '" + - std::string(RVS_CONF_SUBTEST) + "' key value"; + std::string(RVS_CONF_TRIAD) + "' key value"; rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); bsts = false; } @@ -198,6 +257,14 @@ bool mem_action::get_all_mem_config_keys(void) { bsts = false; } + if (property_get_int(RVS_CONF_DURATION, + &duration, MEM_DEFAULT_DURATION)) { + msg = "invalid '" + + std::string(RVS_CONF_DURATION) + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + if (property_get(RVS_CONF_MEM_MIBIBYTE, &mibibytes, MEM_DEFAULT_MEM_MIBIBYTE)) { msg = "invalid '" + @@ -237,6 +304,51 @@ bool mem_action::get_all_mem_config_keys(void) { bsts = false; } + auto it = property.find(RVS_CONF_DATA_INIT); + if (it != property.end()) { + data_init = it->second; + } else { + data_init = MEM_DEFAULT_DATA_INIT; + } + + if (data_init != "default" && data_init != "gpu_norm_dist" && + data_init != "cpu_norm_dist" && data_init != "zero_init") { + msg = "invalid '" + std::string(RVS_CONF_DATA_INIT) + + "' key value '" + data_init + + "'. Must be 'default', 'gpu_norm_dist', 'cpu_norm_dist' or 'zero_init'"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + it = property.find(RVS_CONF_NONTEMPORAL); + if (it != property.end()) { + nontemporal = it->second; + } else { + nontemporal = MEM_DEFAULT_NONTEMPORAL; + } + + if (nontemporal != "none" && nontemporal != "all" && + nontemporal != "read" && nontemporal != "write") { + msg = "invalid '" + std::string(RVS_CONF_NONTEMPORAL) + + "' key value '" + nontemporal + + "'. Must be 'none', 'all', 'read' or 'write'"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + if (num_iterations < 2) { + msg = "invalid '" + + std::string(RVS_CONF_NUM_ITER) + "' key value" + " - expected value greater than 1" ; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + if (property_get(RVS_CONF_SUSTAINED, &sustained, MEM_DEFAULT_SUSTAINED)) { + msg = "invalid '" + std::string(RVS_CONF_SUSTAINED) + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + return bsts; } @@ -270,7 +382,7 @@ int mem_action::get_num_amd_gpu_devices(void) { rvs::lp::AddString(json_root_node, "ERROR", MEM_NO_COMPATIBLE_GPUS); rvs::lp::LogRecordFlush(json_root_node); } - return 0; + return -1; } return hip_num_gpu_devices; } diff --git a/babel.so/src/rvs_memworker.cpp b/babel.so/src/rvs_memworker.cpp index 4ba712a7c..3916aa71d 100644 --- a/babel.so/src/rvs_memworker.cpp +++ b/babel.so/src/rvs_memworker.cpp @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2025 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -39,8 +39,9 @@ using std::string; bool MemWorker::bjson = false; -extern void run_babel(std::pair device, int num_times, int array_size, bool output_csv, bool mibibytes, - int test_type, int subtest, uint16_t dwords_per_lane, uint16_t chunks_per_block, uint16_t tb_size, bool json, std::string action, int rwtest); +extern bool run_babel(std::pair device, int num_times, int array_size, bool output_csv, bool mibibytes, + int test_type, uint16_t dwords_per_lane, uint16_t chunks_per_block, uint16_t tb_size, bool json, std::string action, subtest *test, + const std::string& data_init, const std::string& nontemporal, uint64_t duration, bool sustained); #define FLOAT_TEST 1 #define DOUBLE_TEST 2 @@ -80,8 +81,11 @@ void MemWorker::run() { HIP_CHECK(hipSetDevice(deviceId)); - run_babel(device, num_iterations, array_size, output_csv, mibibytes, - test_type, subtest, dwords_per_lane, chunks_per_block, tb_size, - MemWorker::bjson, action_name, rwtest); + /* Set Babel subtests enable/disable */ + subtest test = {read, write, copy, add, mul, dot, triad}; + + result = run_babel(device, num_iterations, array_size, output_csv, mibibytes, + test_type, dwords_per_lane, chunks_per_block, tb_size, + MemWorker::bjson, action_name, &test, data_init, nontemporal, duration, sustained); } diff --git a/babel.so/src/rvs_module.cpp b/babel.so/src/rvs_module.cpp index 8b9dd0971..034b92438 100644 --- a/babel.so/src/rvs_module.cpp +++ b/babel.so/src/rvs_module.cpp @@ -74,7 +74,6 @@ extern "C" int rvs_module_init(void* pMi) { } extern "C" int rvs_module_terminate(void) { - cleanup_logs(); amdsmi_shut_down(); return 0; } diff --git a/babel.so/src/rvs_stream.cpp b/babel.so/src/rvs_stream.cpp index 797a7f44c..2de3ebd45 100644 --- a/babel.so/src/rvs_stream.cpp +++ b/babel.so/src/rvs_stream.cpp @@ -8,8 +8,11 @@ #include "include/HIPStream.h" #include "hip/hip_runtime.h" #include "include/rvsloglp.h" +#include "hiprand/hiprand.hpp" #include +#include +#include #ifndef TBSIZE #define TBSIZE 1024 @@ -59,6 +62,61 @@ __device__ __forceinline__ constexpr T scalar(const T scalar) { } \ } while(0) +// Non-temporal load/store control +// NT_ALL (1) : NT load + NT store (default) +// NT_READ (2) : NT load + normal store +// NT_WRITE(3) : normal load + NT store +// NT_NONE (0) : normal load + normal store +enum NTMode { NT_NONE = 0, NT_ALL = 1, NT_READ = 2, NT_WRITE = 3 }; + +#define NT_KERNEL_LAUNCH_EVENTS(kernel, epl, cpb, T, grid, block, smem, start, stop, ...) \ + switch (nt_mode) { \ + case NT_NONE: hipLaunchKernelWithEvents((kernel), \ + grid, block, smem, start, stop, __VA_ARGS__); break; \ + case NT_READ: hipLaunchKernelWithEvents((kernel), \ + grid, block, smem, start, stop, __VA_ARGS__); break; \ + case NT_WRITE: hipLaunchKernelWithEvents((kernel), \ + grid, block, smem, start, stop, __VA_ARGS__); break; \ + default: hipLaunchKernelWithEvents((kernel), \ + grid, block, smem, start, stop, __VA_ARGS__); break; \ + } + +#define NT_KERNEL_LAUNCH_SYNC(kernel, epl, cpb, T, grid, block, smem, stop, ...) \ + switch (nt_mode) { \ + case NT_NONE: hipLaunchKernelSynchronous((kernel), \ + grid, block, smem, stop, __VA_ARGS__); break; \ + case NT_READ: hipLaunchKernelSynchronous((kernel), \ + grid, block, smem, stop, __VA_ARGS__); break; \ + case NT_WRITE: hipLaunchKernelSynchronous((kernel), \ + grid, block, smem, stop, __VA_ARGS__); break; \ + default: hipLaunchKernelSynchronous((kernel), \ + grid, block, smem, stop, __VA_ARGS__); break; \ + } + +#define NT_KERNEL_LAUNCH_ASYNC(kernel, epl, cpb, T, grid, block, ...) \ + switch (nt_mode) { \ + case NT_NONE: hipLaunchKernelGGL((kernel),\ + grid, block, 0, nullptr, __VA_ARGS__); break; \ + case NT_READ: hipLaunchKernelGGL((kernel),\ + grid, block, 0, nullptr, __VA_ARGS__); break; \ + case NT_WRITE: hipLaunchKernelGGL((kernel),\ + grid, block, 0, nullptr, __VA_ARGS__); break; \ + default: hipLaunchKernelGGL((kernel),\ + grid, block, 0, nullptr, __VA_ARGS__); break; \ + } + +#define NT_KERNEL_LAUNCH_DOT_SYNC(kernel, epl, cpb, T, tbs, grid, block, smem, stop, ...) \ + switch (nt_mode) { \ + case NT_NONE: hipLaunchKernelSynchronous((kernel), \ + grid, block, smem, stop, __VA_ARGS__); break; \ + case NT_READ: hipLaunchKernelSynchronous((kernel), \ + grid, block, smem, stop, __VA_ARGS__); break; \ + case NT_WRITE: hipLaunchKernelSynchronous((kernel), \ + grid, block, smem, stop, __VA_ARGS__); break; \ + default: hipLaunchKernelSynchronous((kernel), \ + grid, block, smem, stop, __VA_ARGS__); break; \ + } + template static void hipLaunchKernelWithEvents(F kernel, const dim3& numBlocks, const dim3& dimBlocks, hipStream_t stream, @@ -94,12 +152,20 @@ static void hipLaunchKernelSynchronous(F kernel, const dim3& numBlocks, template HIPStream::HIPStream(const unsigned int ARRAY_SIZE, const bool event_timing, const int device_index, const unsigned int _dwords_per_lane, const unsigned int _chunks_per_block, - const unsigned int _threads_per_block) + const unsigned int _threads_per_block, const std::string& nontemporal) : array_size{ARRAY_SIZE}, evt_timing(event_timing), - dwords_per_lane(_dwords_per_lane), chunks_per_block(_chunks_per_block), tb_size(_threads_per_block) + dwords_per_lane(_dwords_per_lane), chunks_per_block(_chunks_per_block), tb_size(_threads_per_block), + sustained_mode(false) { + std::string msg; + // Set Non-temporal mode + if (nontemporal == "none") nt_mode = 0; // NT_NONE + else if (nontemporal == "read") nt_mode = 2; // NT_READ + else if (nontemporal == "write") nt_mode = 3; // NT_WRITE + else nt_mode = 1; // NT_ALL + // make sure that either: // DWORDS_PER_LANE is less than sizeof(T), in which case we default to 1 element // or @@ -178,6 +244,12 @@ HIPStream::~HIPStream() check_error(hipEventDestroy(coherent_ev)); } +template +void HIPStream::sustained_sync() +{ + check_error(hipDeviceSynchronize()); +} + template __global__ void init_kernel(T * a, T * b, T * c, T initA, T initB, T initC) @@ -197,6 +269,98 @@ void HIPStream::init_arrays(T initA, T initB, T initC) check_error(hipDeviceSynchronize()); } +template +void HIPStream::init_arrays_normdist( + T mean, T stddev, bool gpu_init, + std::vector& a, std::vector& b, std::vector& c) +{ + if (!gpu_init) { + +#if !defined(USE_CPU_THREADS_INIT) + rvs::lp::Log("Using a Single Thread on CPU to Initialize NORMAL distributed data", + rvs::loginfo); + + std::random_device rd{}; + std::mt19937_64 gen{rd()}; + std::normal_distribution dist{mean, stddev}; + + auto gen_func = [&]() { return dist(gen); }; + + std::generate(a.begin(), a.end(), gen_func); + std::generate(b.begin(), b.end(), gen_func); + std::generate(c.begin(), c.end(), gen_func); + +#else + constexpr uint32_t NUM_CHUNKS = NUM_CPU_THREADS_INIT; + rvs::lp::Log(std::string("Using ") + std::to_string(NUM_CHUNKS) + + " Threads on CPU to Initialize NORMAL distributed data", + rvs::loginfo); + + std::vector workers; + workers.reserve(NUM_CHUNKS); + + const uint32_t CHUNK_SIZE = array_size / NUM_CHUNKS; + + for (uint32_t work_id = 0; work_id < NUM_CHUNKS; ++work_id) { + const uint32_t start = work_id * CHUNK_SIZE; + const uint32_t end = + (work_id == NUM_CHUNKS - 1) ? array_size : start + CHUNK_SIZE; + + workers.emplace_back([&, start, end]() { + std::random_device rd{}; + std::mt19937_64 gen{rd()}; + std::normal_distribution dist{mean, stddev}; + auto gen_func = [&]() { return dist(gen); }; + + std::generate(a.begin() + start, a.begin() + end, gen_func); + std::generate(b.begin() + start, b.begin() + end, gen_func); + std::generate(c.begin() + start, c.begin() + end, gen_func); + }); + } + + for (auto& w : workers) { + w.join(); + } +#endif + + check_error(hipMemcpy(d_a, a.data(), sizeof(T) * array_size, hipMemcpyHostToDevice)); + check_error(hipMemcpy(d_b, b.data(), sizeof(T) * array_size, hipMemcpyHostToDevice)); + check_error(hipMemcpy(d_c, c.data(), sizeof(T) * array_size, hipMemcpyHostToDevice)); + + } else { + + rvs::lp::Log("WARNING: Using GPU based NORMAL Distribution Initialization\n" + "CPU based NORMAL Distribution Initialization is RECOMMENDED when validating", + rvs::loginfo); + + hiprand_cpp::mt19937_engine engine; + hiprand_cpp::normal_distribution dist{mean, stddev}; + + try { dist(engine, d_a, array_size); } + catch (const std::exception& e) { + std::string err = std::string("hipRAND ERROR: ") + e.what(); + rvs::lp::Log(err, rvs::logerror); + std::exit(EXIT_FAILURE); + } + + try { dist(engine, d_b, array_size); } + catch (const std::exception& e) { + std::string err = std::string("hipRAND ERROR: ") + e.what(); + rvs::lp::Log(err, rvs::logerror); + std::exit(EXIT_FAILURE); + } + + try { dist(engine, d_c, array_size); } + catch (const std::exception& e) { + std::string err = std::string("hipRAND ERROR: ") + e.what(); + rvs::lp::Log(err, rvs::logerror); + std::exit(EXIT_FAILURE); + } + + check_error(hipDeviceSynchronize()); + } +} + template void HIPStream::read_arrays(std::vector& a, std::vector& b, std::vector& c) { @@ -207,35 +371,23 @@ void HIPStream::read_arrays(std::vector& a, std::vector& b, std::vector check_error(hipMemcpy(c.data(), d_c, c.size()*sizeof(T), hipMemcpyDeviceToHost)); } - -// turn on non-temporal by default -#ifndef NONTEMPORAL -#define NONTEMPORAL 1 -#endif - -#if NONTEMPORAL == 0 -template -__device__ __forceinline__ T load(const T& ref) { - return ref; -} - -template -__device__ __forceinline__ void store(const T& value, T& ref) { - ref = value; -} -#else -template +template __device__ __forceinline__ T load(const T& ref) { - return __builtin_nontemporal_load(&ref); + if constexpr (nt_mode == NT_ALL || nt_mode == NT_READ) + return __builtin_nontemporal_load(&ref); + else + return ref; } -template +template __device__ __forceinline__ void store(const T& value, T& ref) { - __builtin_nontemporal_store(value, &ref); + if constexpr (nt_mode == NT_ALL || nt_mode == NT_WRITE) + __builtin_nontemporal_store(value, &ref); + else + ref = value; } -#endif -template +template __launch_bounds__(TBSIZE) __global__ void read_kernel(const T * __restrict a, T * __restrict c) @@ -248,7 +400,7 @@ void read_kernel(const T * __restrict a, T * __restrict c) { for (auto j = 0u; j != elements_per_lane; ++j) { - tmp += load(a[gidx + i * dx + j]); + tmp += load(a[gidx + i * dx + j]); } } @@ -263,37 +415,40 @@ template float HIPStream::read() { float kernel_time = 0.; + if (sustained_mode) + { + if(elements_per_lane == 4 && chunks_per_block == 1) + NT_KERNEL_LAUNCH_ASYNC(read_kernel, 4, 1, T, dim3(block_cnt), dim3(tb_size), d_a, d_c) + else if(elements_per_lane == 2 && chunks_per_block == 1) + NT_KERNEL_LAUNCH_ASYNC(read_kernel, 2, 1, T, dim3(block_cnt), dim3(tb_size), d_a, d_c) + else if(elements_per_lane == 4 && chunks_per_block == 2) + NT_KERNEL_LAUNCH_ASYNC(read_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), d_a, d_c) + else if(elements_per_lane == 2 && chunks_per_block == 2) + NT_KERNEL_LAUNCH_ASYNC(read_kernel, 2, 2, T, dim3(block_cnt), dim3(tb_size), d_a, d_c) + else if(elements_per_lane == 4 && chunks_per_block == 4) + NT_KERNEL_LAUNCH_ASYNC(read_kernel, 4, 4, T, dim3(block_cnt), dim3(tb_size), d_a, d_c) + else if(elements_per_lane == 2 && chunks_per_block == 4) + NT_KERNEL_LAUNCH_ASYNC(read_kernel, 2, 4, T, dim3(block_cnt), dim3(tb_size), d_a, d_c) + else + NT_KERNEL_LAUNCH_ASYNC(read_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), d_a, d_c) + return 0.f; + } if (evt_timing) { - // Support for dwords_per_lane = 4 & chunks_per_block = 1/2/4 if(elements_per_lane == 4 && chunks_per_block == 1) - hipLaunchKernelWithEvents(read_kernel<4, 1, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_c); + NT_KERNEL_LAUNCH_EVENTS(read_kernel, 4, 1, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_c) else if(elements_per_lane == 2 && chunks_per_block == 1) - hipLaunchKernelWithEvents(read_kernel<2, 1, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_c); + NT_KERNEL_LAUNCH_EVENTS(read_kernel, 2, 1, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_c) else if(elements_per_lane == 4 && chunks_per_block == 2) - hipLaunchKernelWithEvents(read_kernel<4, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_c); + NT_KERNEL_LAUNCH_EVENTS(read_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_c) else if(elements_per_lane == 2 && chunks_per_block == 2) - hipLaunchKernelWithEvents(read_kernel<2, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_c); + NT_KERNEL_LAUNCH_EVENTS(read_kernel, 2, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_c) else if(elements_per_lane == 4 && chunks_per_block == 4) - hipLaunchKernelWithEvents(read_kernel<4, 4, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_c); + NT_KERNEL_LAUNCH_EVENTS(read_kernel, 4, 4, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_c) else if(elements_per_lane == 2 && chunks_per_block == 4) - hipLaunchKernelWithEvents(read_kernel<2, 4, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_c); + NT_KERNEL_LAUNCH_EVENTS(read_kernel, 2, 4, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_c) else - hipLaunchKernelWithEvents(read_kernel<4, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_c); + NT_KERNEL_LAUNCH_EVENTS(read_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_c) check_error(hipEventSynchronize(stop_ev)); check_error(hipEventElapsedTime(&kernel_time, start_ev, stop_ev)); @@ -301,38 +456,24 @@ float HIPStream::read() else { if(elements_per_lane == 4 && chunks_per_block == 1) - hipLaunchKernelSynchronous(read_kernel<4, 1, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_c); + NT_KERNEL_LAUNCH_SYNC(read_kernel, 4, 1, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_c) else if(elements_per_lane == 2 && chunks_per_block == 1) - hipLaunchKernelSynchronous(read_kernel<2, 1, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_c); + NT_KERNEL_LAUNCH_SYNC(read_kernel, 2, 1, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_c) else if(elements_per_lane == 4 && chunks_per_block == 2) - hipLaunchKernelSynchronous(read_kernel<4, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_c); + NT_KERNEL_LAUNCH_SYNC(read_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_c) else if(elements_per_lane == 2 && chunks_per_block == 2) - hipLaunchKernelSynchronous(read_kernel<2, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_c); + NT_KERNEL_LAUNCH_SYNC(read_kernel, 2, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_c) else if(elements_per_lane == 4 && chunks_per_block == 4) - hipLaunchKernelSynchronous(read_kernel<4, 4, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_c); + NT_KERNEL_LAUNCH_SYNC(read_kernel, 4, 4, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_c) else if(elements_per_lane == 2 && chunks_per_block == 4) - hipLaunchKernelSynchronous(read_kernel<2, 4, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_c); + NT_KERNEL_LAUNCH_SYNC(read_kernel, 2, 4, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_c) else - hipLaunchKernelSynchronous(read_kernel<4, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_c); + NT_KERNEL_LAUNCH_SYNC(read_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_c) } return kernel_time; } -template +template __launch_bounds__(TBSIZE) __global__ void write_kernel(T * __restrict c) @@ -344,7 +485,7 @@ void write_kernel(T * __restrict c) { for (auto j = 0u; j != elements_per_lane; ++j) { - store(scalar(startC), c[gidx + i * dx + j]); + store(scalar(startC), c[gidx + i * dx + j]); } } } @@ -353,37 +494,40 @@ template float HIPStream::write() { float kernel_time = 0.; + if (sustained_mode) + { + if(elements_per_lane == 4 && chunks_per_block == 1) + NT_KERNEL_LAUNCH_ASYNC(write_kernel, 4, 1, T, dim3(block_cnt), dim3(tb_size), d_c) + else if(elements_per_lane == 2 && chunks_per_block == 1) + NT_KERNEL_LAUNCH_ASYNC(write_kernel, 2, 1, T, dim3(block_cnt), dim3(tb_size), d_c) + else if(elements_per_lane == 4 && chunks_per_block == 2) + NT_KERNEL_LAUNCH_ASYNC(write_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), d_c) + else if(elements_per_lane == 2 && chunks_per_block == 2) + NT_KERNEL_LAUNCH_ASYNC(write_kernel, 2, 2, T, dim3(block_cnt), dim3(tb_size), d_c) + else if(elements_per_lane == 4 && chunks_per_block == 4) + NT_KERNEL_LAUNCH_ASYNC(write_kernel, 4, 4, T, dim3(block_cnt), dim3(tb_size), d_c) + else if(elements_per_lane == 2 && chunks_per_block == 4) + NT_KERNEL_LAUNCH_ASYNC(write_kernel, 2, 4, T, dim3(block_cnt), dim3(tb_size), d_c) + else + NT_KERNEL_LAUNCH_ASYNC(write_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), d_c) + return 0.f; + } if (evt_timing) { - // Support for dwords_per_lane = 4 & chunks_per_block = 1/2/4 if(elements_per_lane == 4 && chunks_per_block == 1) - hipLaunchKernelWithEvents(write_kernel<4, 1, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_c); + NT_KERNEL_LAUNCH_EVENTS(write_kernel, 4, 1, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_c) else if(elements_per_lane == 2 && chunks_per_block == 1) - hipLaunchKernelWithEvents(write_kernel<2, 1, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_c); + NT_KERNEL_LAUNCH_EVENTS(write_kernel, 2, 1, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_c) else if(elements_per_lane == 4 && chunks_per_block == 2) - hipLaunchKernelWithEvents(write_kernel<4, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_c); + NT_KERNEL_LAUNCH_EVENTS(write_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_c) else if(elements_per_lane == 2 && chunks_per_block == 2) - hipLaunchKernelWithEvents(write_kernel<2, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_c); + NT_KERNEL_LAUNCH_EVENTS(write_kernel, 2, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_c) else if(elements_per_lane == 4 && chunks_per_block == 4) - hipLaunchKernelWithEvents(write_kernel<4, 4, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_c); + NT_KERNEL_LAUNCH_EVENTS(write_kernel, 4, 4, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_c) else if(elements_per_lane == 2 && chunks_per_block == 4) - hipLaunchKernelWithEvents(write_kernel<2, 4, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_c); + NT_KERNEL_LAUNCH_EVENTS(write_kernel, 2, 4, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_c) else - hipLaunchKernelWithEvents(write_kernel<4, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_c); + NT_KERNEL_LAUNCH_EVENTS(write_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_c) check_error(hipEventSynchronize(stop_ev)); check_error(hipEventElapsedTime(&kernel_time, start_ev, stop_ev)); @@ -391,38 +535,24 @@ float HIPStream::write() else { if(elements_per_lane == 4 && chunks_per_block == 1) - hipLaunchKernelSynchronous(write_kernel<4, 1, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_c); + NT_KERNEL_LAUNCH_SYNC(write_kernel, 4, 1, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_c) else if(elements_per_lane == 2 && chunks_per_block == 1) - hipLaunchKernelSynchronous(write_kernel<2, 1, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_c); + NT_KERNEL_LAUNCH_SYNC(write_kernel, 2, 1, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_c) else if(elements_per_lane == 4 && chunks_per_block == 2) - hipLaunchKernelSynchronous(write_kernel<4, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_c); + NT_KERNEL_LAUNCH_SYNC(write_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_c) else if(elements_per_lane == 2 && chunks_per_block == 2) - hipLaunchKernelSynchronous(write_kernel<2, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_c); + NT_KERNEL_LAUNCH_SYNC(write_kernel, 2, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_c) else if(elements_per_lane == 4 && chunks_per_block == 4) - hipLaunchKernelSynchronous(write_kernel<4, 4, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_c); + NT_KERNEL_LAUNCH_SYNC(write_kernel, 4, 4, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_c) else if(elements_per_lane == 2 && chunks_per_block == 4) - hipLaunchKernelSynchronous(write_kernel<2, 4, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_c); + NT_KERNEL_LAUNCH_SYNC(write_kernel, 2, 4, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_c) else - hipLaunchKernelSynchronous(write_kernel<4, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_c); + NT_KERNEL_LAUNCH_SYNC(write_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_c) } return kernel_time; } -template +template __launch_bounds__(TBSIZE) __global__ void copy_kernel(const T * __restrict a, T * __restrict c) @@ -434,7 +564,7 @@ void copy_kernel(const T * __restrict a, T * __restrict c) { for (auto j = 0u; j != elements_per_lane; ++j) { - store(load(a[gidx + i * dx + j]), c[gidx + i * dx + j]); + store(load(a[gidx + i * dx + j]), c[gidx + i * dx + j]); } } } @@ -443,37 +573,40 @@ template float HIPStream::copy() { float kernel_time = 0.; + if (sustained_mode) + { + if(elements_per_lane == 4 && chunks_per_block == 1) + NT_KERNEL_LAUNCH_ASYNC(copy_kernel, 4, 1, T, dim3(block_cnt), dim3(tb_size), d_a, d_c) + else if(elements_per_lane == 2 && chunks_per_block == 1) + NT_KERNEL_LAUNCH_ASYNC(copy_kernel, 2, 1, T, dim3(block_cnt), dim3(tb_size), d_a, d_c) + else if(elements_per_lane == 4 && chunks_per_block == 2) + NT_KERNEL_LAUNCH_ASYNC(copy_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), d_a, d_c) + else if(elements_per_lane == 2 && chunks_per_block == 2) + NT_KERNEL_LAUNCH_ASYNC(copy_kernel, 2, 2, T, dim3(block_cnt), dim3(tb_size), d_a, d_c) + else if(elements_per_lane == 4 && chunks_per_block == 4) + NT_KERNEL_LAUNCH_ASYNC(copy_kernel, 4, 4, T, dim3(block_cnt), dim3(tb_size), d_a, d_c) + else if(elements_per_lane == 2 && chunks_per_block == 4) + NT_KERNEL_LAUNCH_ASYNC(copy_kernel, 2, 4, T, dim3(block_cnt), dim3(tb_size), d_a, d_c) + else + NT_KERNEL_LAUNCH_ASYNC(copy_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), d_a, d_c) + return 0.f; + } if (evt_timing) { - // Support for dwords_per_lane = 4 & chunks_per_block = 1/2/4 if(elements_per_lane == 4 && chunks_per_block == 1) - hipLaunchKernelWithEvents(copy_kernel<4, 1, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_c); + NT_KERNEL_LAUNCH_EVENTS(copy_kernel, 4, 1, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_c) else if(elements_per_lane == 2 && chunks_per_block == 1) - hipLaunchKernelWithEvents(copy_kernel<2, 1, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_c); + NT_KERNEL_LAUNCH_EVENTS(copy_kernel, 2, 1, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_c) else if(elements_per_lane == 4 && chunks_per_block == 2) - hipLaunchKernelWithEvents(copy_kernel<4, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_c); + NT_KERNEL_LAUNCH_EVENTS(copy_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_c) else if(elements_per_lane == 2 && chunks_per_block == 2) - hipLaunchKernelWithEvents(copy_kernel<2, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_c); + NT_KERNEL_LAUNCH_EVENTS(copy_kernel, 2, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_c) else if(elements_per_lane == 4 && chunks_per_block == 4) - hipLaunchKernelWithEvents(copy_kernel<4, 4, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_c); + NT_KERNEL_LAUNCH_EVENTS(copy_kernel, 4, 4, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_c) else if(elements_per_lane == 2 && chunks_per_block == 4) - hipLaunchKernelWithEvents(copy_kernel<2, 4, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_c); + NT_KERNEL_LAUNCH_EVENTS(copy_kernel, 2, 4, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_c) else - hipLaunchKernelWithEvents(copy_kernel<4, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_c); + NT_KERNEL_LAUNCH_EVENTS(copy_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_c) check_error(hipEventSynchronize(stop_ev)); check_error(hipEventElapsedTime(&kernel_time, start_ev, stop_ev)); @@ -481,38 +614,24 @@ float HIPStream::copy() else { if(elements_per_lane == 4 && chunks_per_block == 1) - hipLaunchKernelSynchronous(copy_kernel<4, 1, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_c); + NT_KERNEL_LAUNCH_SYNC(copy_kernel, 4, 1, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_c) else if(elements_per_lane == 2 && chunks_per_block == 1) - hipLaunchKernelSynchronous(copy_kernel<2, 1, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_c); + NT_KERNEL_LAUNCH_SYNC(copy_kernel, 2, 1, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_c) else if(elements_per_lane == 4 && chunks_per_block == 2) - hipLaunchKernelSynchronous(copy_kernel<4, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_c); + NT_KERNEL_LAUNCH_SYNC(copy_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_c) else if(elements_per_lane == 2 && chunks_per_block == 2) - hipLaunchKernelSynchronous(copy_kernel<2, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_c); + NT_KERNEL_LAUNCH_SYNC(copy_kernel, 2, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_c) else if(elements_per_lane == 4 && chunks_per_block == 4) - hipLaunchKernelSynchronous(copy_kernel<4, 4, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_c); + NT_KERNEL_LAUNCH_SYNC(copy_kernel, 4, 4, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_c) else if(elements_per_lane == 2 && chunks_per_block == 4) - hipLaunchKernelSynchronous(copy_kernel<2, 4, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_c); + NT_KERNEL_LAUNCH_SYNC(copy_kernel, 2, 4, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_c) else - hipLaunchKernelSynchronous(copy_kernel<4, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_c); + NT_KERNEL_LAUNCH_SYNC(copy_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_c) } return kernel_time; } -template +template __launch_bounds__(TBSIZE) __global__ void mul_kernel(T * __restrict b, const T * __restrict c) @@ -524,7 +643,7 @@ void mul_kernel(T * __restrict b, const T * __restrict c) { for (auto j = 0u; j != elements_per_lane; ++j) { - store(scalar(startScalar) * load(c[gidx + i * dx + j]), b[gidx + i * dx + j]); + store(scalar(startScalar) * load(c[gidx + i * dx + j]), b[gidx + i * dx + j]); } } } @@ -533,37 +652,40 @@ template float HIPStream::mul() { float kernel_time = 0.; + if (sustained_mode) + { + if(elements_per_lane == 4 && chunks_per_block == 1) + NT_KERNEL_LAUNCH_ASYNC(mul_kernel, 4, 1, T, dim3(block_cnt), dim3(tb_size), d_b, d_c) + else if(elements_per_lane == 2 && chunks_per_block == 1) + NT_KERNEL_LAUNCH_ASYNC(mul_kernel, 2, 1, T, dim3(block_cnt), dim3(tb_size), d_b, d_c) + else if(elements_per_lane == 4 && chunks_per_block == 2) + NT_KERNEL_LAUNCH_ASYNC(mul_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), d_b, d_c) + else if(elements_per_lane == 2 && chunks_per_block == 2) + NT_KERNEL_LAUNCH_ASYNC(mul_kernel, 2, 2, T, dim3(block_cnt), dim3(tb_size), d_b, d_c) + else if(elements_per_lane == 4 && chunks_per_block == 4) + NT_KERNEL_LAUNCH_ASYNC(mul_kernel, 4, 4, T, dim3(block_cnt), dim3(tb_size), d_b, d_c) + else if(elements_per_lane == 2 && chunks_per_block == 4) + NT_KERNEL_LAUNCH_ASYNC(mul_kernel, 2, 4, T, dim3(block_cnt), dim3(tb_size), d_b, d_c) + else + NT_KERNEL_LAUNCH_ASYNC(mul_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), d_b, d_c) + return 0.f; + } if (evt_timing) { - // Support for dwords_per_lane = 4 & chunks_per_block = 1/2/4 if(elements_per_lane == 4 && chunks_per_block == 1) - hipLaunchKernelWithEvents(mul_kernel<4, 1, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_b, d_c); + NT_KERNEL_LAUNCH_EVENTS(mul_kernel, 4, 1, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_b, d_c) else if(elements_per_lane == 2 && chunks_per_block == 1) - hipLaunchKernelWithEvents(mul_kernel<2, 1, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_b, d_c); + NT_KERNEL_LAUNCH_EVENTS(mul_kernel, 2, 1, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_b, d_c) else if(elements_per_lane == 4 && chunks_per_block == 2) - hipLaunchKernelWithEvents(mul_kernel<4, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_b, d_c); + NT_KERNEL_LAUNCH_EVENTS(mul_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_b, d_c) else if(elements_per_lane == 2 && chunks_per_block == 2) - hipLaunchKernelWithEvents(mul_kernel<2, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_b, d_c); + NT_KERNEL_LAUNCH_EVENTS(mul_kernel, 2, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_b, d_c) else if(elements_per_lane == 4 && chunks_per_block == 4) - hipLaunchKernelWithEvents(mul_kernel<4, 4, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_b, d_c); + NT_KERNEL_LAUNCH_EVENTS(mul_kernel, 4, 4, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_b, d_c) else if(elements_per_lane == 2 && chunks_per_block == 4) - hipLaunchKernelWithEvents(mul_kernel<2, 4, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_b, d_c); + NT_KERNEL_LAUNCH_EVENTS(mul_kernel, 2, 4, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_b, d_c) else - hipLaunchKernelWithEvents(mul_kernel<4, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_b, d_c); + NT_KERNEL_LAUNCH_EVENTS(mul_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_b, d_c) check_error(hipEventSynchronize(stop_ev)); check_error(hipEventElapsedTime(&kernel_time, start_ev, stop_ev)); @@ -571,38 +693,24 @@ float HIPStream::mul() else { if(elements_per_lane == 4 && chunks_per_block == 1) - hipLaunchKernelSynchronous(mul_kernel<4, 1, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_b, d_c); + NT_KERNEL_LAUNCH_SYNC(mul_kernel, 4, 1, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_b, d_c) else if(elements_per_lane == 2 && chunks_per_block == 1) - hipLaunchKernelSynchronous(mul_kernel<2, 1, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_b, d_c); + NT_KERNEL_LAUNCH_SYNC(mul_kernel, 2, 1, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_b, d_c) else if(elements_per_lane == 4 && chunks_per_block == 2) - hipLaunchKernelSynchronous(mul_kernel<4, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_b, d_c); + NT_KERNEL_LAUNCH_SYNC(mul_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_b, d_c) else if(elements_per_lane == 2 && chunks_per_block == 2) - hipLaunchKernelSynchronous(mul_kernel<2, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_b, d_c); + NT_KERNEL_LAUNCH_SYNC(mul_kernel, 2, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_b, d_c) else if(elements_per_lane == 4 && chunks_per_block == 4) - hipLaunchKernelSynchronous(mul_kernel<4, 4, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_b, d_c); + NT_KERNEL_LAUNCH_SYNC(mul_kernel, 4, 4, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_b, d_c) else if(elements_per_lane == 2 && chunks_per_block == 4) - hipLaunchKernelSynchronous(mul_kernel<2, 4, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_b, d_c); + NT_KERNEL_LAUNCH_SYNC(mul_kernel, 2, 4, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_b, d_c) else - hipLaunchKernelSynchronous(mul_kernel<4, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_b, d_c); + NT_KERNEL_LAUNCH_SYNC(mul_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_b, d_c) } return kernel_time; } -template +template __launch_bounds__(TBSIZE) __global__ void add_kernel(const T * __restrict a, const T * __restrict b, @@ -615,7 +723,7 @@ void add_kernel(const T * __restrict a, const T * __restrict b, { for (auto j = 0u; j != elements_per_lane; ++j) { - store(load(a[gidx + i * dx + j]) + load(b[gidx + i * dx + j]), c[gidx + i * dx + j]); + store(load(a[gidx + i * dx + j]) + load(b[gidx + i * dx + j]), c[gidx + i * dx + j]); } } } @@ -624,37 +732,40 @@ template float HIPStream::add() { float kernel_time = 0.; + if (sustained_mode) + { + if(elements_per_lane == 4 && chunks_per_block == 1) + NT_KERNEL_LAUNCH_ASYNC(add_kernel, 4, 1, T, dim3(block_cnt), dim3(tb_size), d_a, d_b, d_c) + else if(elements_per_lane == 2 && chunks_per_block == 1) + NT_KERNEL_LAUNCH_ASYNC(add_kernel, 2, 1, T, dim3(block_cnt), dim3(tb_size), d_a, d_b, d_c) + else if(elements_per_lane == 4 && chunks_per_block == 2) + NT_KERNEL_LAUNCH_ASYNC(add_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), d_a, d_b, d_c) + else if(elements_per_lane == 2 && chunks_per_block == 2) + NT_KERNEL_LAUNCH_ASYNC(add_kernel, 2, 2, T, dim3(block_cnt), dim3(tb_size), d_a, d_b, d_c) + else if(elements_per_lane == 4 && chunks_per_block == 4) + NT_KERNEL_LAUNCH_ASYNC(add_kernel, 4, 4, T, dim3(block_cnt), dim3(tb_size), d_a, d_b, d_c) + else if(elements_per_lane == 2 && chunks_per_block == 4) + NT_KERNEL_LAUNCH_ASYNC(add_kernel, 2, 4, T, dim3(block_cnt), dim3(tb_size), d_a, d_b, d_c) + else + NT_KERNEL_LAUNCH_ASYNC(add_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), d_a, d_b, d_c) + return 0.f; + } if (evt_timing) { - // Support for dwords_per_lane = 4 & chunks_per_block = 1/2/4 if(elements_per_lane == 4 && chunks_per_block == 1) - hipLaunchKernelWithEvents(add_kernel<4, 1, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_b, d_c); + NT_KERNEL_LAUNCH_EVENTS(add_kernel, 4, 1, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_b, d_c) else if(elements_per_lane == 2 && chunks_per_block == 1) - hipLaunchKernelWithEvents(add_kernel<2, 1, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_b, d_c); + NT_KERNEL_LAUNCH_EVENTS(add_kernel, 2, 1, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_b, d_c) else if(elements_per_lane == 4 && chunks_per_block == 2) - hipLaunchKernelWithEvents(add_kernel<4, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_b, d_c); + NT_KERNEL_LAUNCH_EVENTS(add_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_b, d_c) else if(elements_per_lane == 2 && chunks_per_block == 2) - hipLaunchKernelWithEvents(add_kernel<2, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_b, d_c); + NT_KERNEL_LAUNCH_EVENTS(add_kernel, 2, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_b, d_c) else if(elements_per_lane == 4 && chunks_per_block == 4) - hipLaunchKernelWithEvents(add_kernel<4, 4, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_b, d_c); + NT_KERNEL_LAUNCH_EVENTS(add_kernel, 4, 4, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_b, d_c) else if(elements_per_lane == 2 && chunks_per_block == 4) - hipLaunchKernelWithEvents(add_kernel<2, 4, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_b, d_c); + NT_KERNEL_LAUNCH_EVENTS(add_kernel, 2, 4, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_b, d_c) else - hipLaunchKernelWithEvents(add_kernel<4, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_b, d_c); + NT_KERNEL_LAUNCH_EVENTS(add_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_b, d_c) check_error(hipEventSynchronize(stop_ev)); check_error(hipEventElapsedTime(&kernel_time, start_ev, stop_ev)); @@ -662,38 +773,24 @@ float HIPStream::add() else { if(elements_per_lane == 4 && chunks_per_block == 1) - hipLaunchKernelSynchronous(add_kernel<4, 1, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_b, d_c); + NT_KERNEL_LAUNCH_SYNC(add_kernel, 4, 1, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_b, d_c) else if(elements_per_lane == 2 && chunks_per_block == 1) - hipLaunchKernelSynchronous(add_kernel<2, 1, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_b, d_c); + NT_KERNEL_LAUNCH_SYNC(add_kernel, 2, 1, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_b, d_c) else if(elements_per_lane == 4 && chunks_per_block == 2) - hipLaunchKernelSynchronous(add_kernel<4, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_b, d_c); + NT_KERNEL_LAUNCH_SYNC(add_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_b, d_c) else if(elements_per_lane == 2 && chunks_per_block == 2) - hipLaunchKernelSynchronous(add_kernel<2, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_b, d_c); + NT_KERNEL_LAUNCH_SYNC(add_kernel, 2, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_b, d_c) else if(elements_per_lane == 4 && chunks_per_block == 4) - hipLaunchKernelSynchronous(add_kernel<4, 4, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_b, d_c); + NT_KERNEL_LAUNCH_SYNC(add_kernel, 4, 4, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_b, d_c) else if(elements_per_lane == 2 && chunks_per_block == 4) - hipLaunchKernelSynchronous(add_kernel<2, 4, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_b, d_c); + NT_KERNEL_LAUNCH_SYNC(add_kernel, 2, 4, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_b, d_c) else - hipLaunchKernelSynchronous(add_kernel<4, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_b, d_c); + NT_KERNEL_LAUNCH_SYNC(add_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_b, d_c) } return kernel_time; } -template +template __launch_bounds__(TBSIZE) __global__ void triad_kernel(T * __restrict a, const T * __restrict b, @@ -706,7 +803,7 @@ void triad_kernel(T * __restrict a, const T * __restrict b, { for (auto j = 0u; j != elements_per_lane; ++j) { - store(load(b[gidx + i * dx + j]) + scalar(startScalar) * load(c[gidx + i * dx + j]), + store(load(b[gidx + i * dx + j]) + scalar(startScalar) * load(c[gidx + i * dx + j]), a[gidx + i * dx + j]); } } @@ -716,37 +813,40 @@ template float HIPStream::triad() { float kernel_time = 0.; + if (sustained_mode) + { + if(elements_per_lane == 4 && chunks_per_block == 1) + NT_KERNEL_LAUNCH_ASYNC(triad_kernel, 4, 1, T, dim3(block_cnt), dim3(tb_size), d_a, d_b, d_c) + else if(elements_per_lane == 2 && chunks_per_block == 1) + NT_KERNEL_LAUNCH_ASYNC(triad_kernel, 2, 1, T, dim3(block_cnt), dim3(tb_size), d_a, d_b, d_c) + else if(elements_per_lane == 4 && chunks_per_block == 2) + NT_KERNEL_LAUNCH_ASYNC(triad_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), d_a, d_b, d_c) + else if(elements_per_lane == 2 && chunks_per_block == 2) + NT_KERNEL_LAUNCH_ASYNC(triad_kernel, 2, 2, T, dim3(block_cnt), dim3(tb_size), d_a, d_b, d_c) + else if(elements_per_lane == 4 && chunks_per_block == 4) + NT_KERNEL_LAUNCH_ASYNC(triad_kernel, 4, 4, T, dim3(block_cnt), dim3(tb_size), d_a, d_b, d_c) + else if(elements_per_lane == 2 && chunks_per_block == 4) + NT_KERNEL_LAUNCH_ASYNC(triad_kernel, 2, 4, T, dim3(block_cnt), dim3(tb_size), d_a, d_b, d_c) + else + NT_KERNEL_LAUNCH_ASYNC(triad_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), d_a, d_b, d_c) + return 0.f; + } if (evt_timing) { - // Support for dwords_per_lane = 4 & chunks_per_block = 1/2/4 if(elements_per_lane == 4 && chunks_per_block == 1) - hipLaunchKernelWithEvents(triad_kernel<4, 1, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_b, d_c); + NT_KERNEL_LAUNCH_EVENTS(triad_kernel, 4, 1, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_b, d_c) else if(elements_per_lane == 2 && chunks_per_block == 1) - hipLaunchKernelWithEvents(triad_kernel<2, 1, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_b, d_c); + NT_KERNEL_LAUNCH_EVENTS(triad_kernel, 2, 1, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_b, d_c) else if(elements_per_lane == 4 && chunks_per_block == 2) - hipLaunchKernelWithEvents(triad_kernel<4, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_b, d_c); + NT_KERNEL_LAUNCH_EVENTS(triad_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_b, d_c) else if(elements_per_lane == 2 && chunks_per_block == 2) - hipLaunchKernelWithEvents(triad_kernel<2, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_b, d_c); + NT_KERNEL_LAUNCH_EVENTS(triad_kernel, 2, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_b, d_c) else if(elements_per_lane == 4 && chunks_per_block == 4) - hipLaunchKernelWithEvents(triad_kernel<4, 4, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_b, d_c); + NT_KERNEL_LAUNCH_EVENTS(triad_kernel, 4, 4, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_b, d_c) else if(elements_per_lane == 2 && chunks_per_block == 4) - hipLaunchKernelWithEvents(triad_kernel<2, 4, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_b, d_c); + NT_KERNEL_LAUNCH_EVENTS(triad_kernel, 2, 4, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_b, d_c) else - hipLaunchKernelWithEvents(triad_kernel<4, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, start_ev, - stop_ev, d_a, d_b, d_c); + NT_KERNEL_LAUNCH_EVENTS(triad_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, start_ev, stop_ev, d_a, d_b, d_c) check_error(hipEventSynchronize(stop_ev)); check_error(hipEventElapsedTime(&kernel_time, start_ev, stop_ev)); @@ -754,33 +854,19 @@ float HIPStream::triad() else { if(elements_per_lane == 4 && chunks_per_block == 1) - hipLaunchKernelSynchronous(triad_kernel<4, 1, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_b, d_c); + NT_KERNEL_LAUNCH_SYNC(triad_kernel, 4, 1, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_b, d_c) else if(elements_per_lane == 2 && chunks_per_block == 1) - hipLaunchKernelSynchronous(triad_kernel<2, 1, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_b, d_c); + NT_KERNEL_LAUNCH_SYNC(triad_kernel, 2, 1, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_b, d_c) else if(elements_per_lane == 4 && chunks_per_block == 2) - hipLaunchKernelSynchronous(triad_kernel<4, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_b, d_c); + NT_KERNEL_LAUNCH_SYNC(triad_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_b, d_c) else if(elements_per_lane == 2 && chunks_per_block == 2) - hipLaunchKernelSynchronous(triad_kernel<2, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_b, d_c); + NT_KERNEL_LAUNCH_SYNC(triad_kernel, 2, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_b, d_c) else if(elements_per_lane == 4 && chunks_per_block == 4) - hipLaunchKernelSynchronous(triad_kernel<4, 4, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_b, d_c); + NT_KERNEL_LAUNCH_SYNC(triad_kernel, 4, 4, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_b, d_c) else if(elements_per_lane == 2 && chunks_per_block == 4) - hipLaunchKernelSynchronous(triad_kernel<2, 4, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_b, d_c); + NT_KERNEL_LAUNCH_SYNC(triad_kernel, 2, 4, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_b, d_c) else - hipLaunchKernelSynchronous(triad_kernel<4, 2, T>, - dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, - d_a, d_b, d_c); + NT_KERNEL_LAUNCH_SYNC(triad_kernel, 4, 2, T, dim3(block_cnt), dim3(tb_size), nullptr, stop_ev, d_a, d_b, d_c) } return kernel_time; } @@ -817,7 +903,7 @@ struct Reducer<1u> { {} }; -template +template __launch_bounds__(TBSIZE) __global__ void dot_kernel(const T * __restrict a, const T * __restrict b, @@ -831,7 +917,7 @@ void dot_kernel(const T * __restrict a, const T * __restrict b, { for (auto j = 0u; j != elements_per_lane; ++j) { - tmp += load(a[gidx + i * dx + j]) * load(b[gidx + i * dx + j]); + tmp += load(a[gidx + i * dx + j]) * load(b[gidx + i * dx + j]); } } @@ -846,89 +932,62 @@ void dot_kernel(const T * __restrict a, const T * __restrict b, { return; } - store(tb_sum[0], sum[blockIdx.x]); + store(tb_sum[0], sum[blockIdx.x]); } template T HIPStream::dot() { - // Support for dwords_per_lane = 4 & chunks_per_block = 1/2/4 if(elements_per_lane == 4 && chunks_per_block == 1) { if(tb_size == 1024) { - hipLaunchKernelSynchronous(dot_kernel<4, 1, T, 1024>, - dim3(block_cnt), dim3(tb_size), nullptr, coherent_ev, - d_a, d_b, sums); + NT_KERNEL_LAUNCH_DOT_SYNC(dot_kernel, 4, 1, T, 1024, dim3(block_cnt), dim3(tb_size), nullptr, coherent_ev, d_a, d_b, sums) } else if(tb_size == 512) { - hipLaunchKernelSynchronous(dot_kernel<4, 1, T, 512>, - dim3(block_cnt), dim3(tb_size), nullptr, coherent_ev, - d_a, d_b, sums); + NT_KERNEL_LAUNCH_DOT_SYNC(dot_kernel, 4, 1, T, 512, dim3(block_cnt), dim3(tb_size), nullptr, coherent_ev, d_a, d_b, sums) } } else if(elements_per_lane == 2 && chunks_per_block == 1) { if(tb_size == 1024) { - hipLaunchKernelSynchronous(dot_kernel<2, 1, T, 1024>, - dim3(block_cnt), dim3(tb_size), nullptr, coherent_ev, - d_a, d_b, sums); + NT_KERNEL_LAUNCH_DOT_SYNC(dot_kernel, 2, 1, T, 1024, dim3(block_cnt), dim3(tb_size), nullptr, coherent_ev, d_a, d_b, sums) } else if(tb_size == 512) { - hipLaunchKernelSynchronous(dot_kernel<2, 1, T, 512>, - dim3(block_cnt), dim3(tb_size), nullptr, coherent_ev, - d_a, d_b, sums); + NT_KERNEL_LAUNCH_DOT_SYNC(dot_kernel, 2, 1, T, 512, dim3(block_cnt), dim3(tb_size), nullptr, coherent_ev, d_a, d_b, sums) } } else if(elements_per_lane == 4 && chunks_per_block == 2) { if(tb_size == 1024) { - hipLaunchKernelSynchronous(dot_kernel<4, 2, T, 1024>, - dim3(block_cnt), dim3(tb_size), nullptr, coherent_ev, - d_a, d_b, sums); + NT_KERNEL_LAUNCH_DOT_SYNC(dot_kernel, 4, 2, T, 1024, dim3(block_cnt), dim3(tb_size), nullptr, coherent_ev, d_a, d_b, sums) } else if(tb_size == 512) { - hipLaunchKernelSynchronous(dot_kernel<4, 2, T, 512>, - dim3(block_cnt), dim3(tb_size), nullptr, coherent_ev, - d_a, d_b, sums); + NT_KERNEL_LAUNCH_DOT_SYNC(dot_kernel, 4, 2, T, 512, dim3(block_cnt), dim3(tb_size), nullptr, coherent_ev, d_a, d_b, sums) } } else if(elements_per_lane == 2 && chunks_per_block == 2) { if(tb_size == 1024) { - hipLaunchKernelSynchronous(dot_kernel<2, 2, T, 1024>, - dim3(block_cnt), dim3(tb_size), nullptr, coherent_ev, - d_a, d_b, sums); + NT_KERNEL_LAUNCH_DOT_SYNC(dot_kernel, 2, 2, T, 1024, dim3(block_cnt), dim3(tb_size), nullptr, coherent_ev, d_a, d_b, sums) } else if(tb_size == 512) { - hipLaunchKernelSynchronous(dot_kernel<2, 2, T, 512>, - dim3(block_cnt), dim3(tb_size), nullptr, coherent_ev, - d_a, d_b, sums); + NT_KERNEL_LAUNCH_DOT_SYNC(dot_kernel, 2, 2, T, 512, dim3(block_cnt), dim3(tb_size), nullptr, coherent_ev, d_a, d_b, sums) } } else if(elements_per_lane == 4 && chunks_per_block == 4) { if(tb_size == 1024) { - hipLaunchKernelSynchronous(dot_kernel<4, 4, T, 1024>, - dim3(block_cnt), dim3(tb_size), nullptr, coherent_ev, - d_a, d_b, sums); + NT_KERNEL_LAUNCH_DOT_SYNC(dot_kernel, 4, 4, T, 1024, dim3(block_cnt), dim3(tb_size), nullptr, coherent_ev, d_a, d_b, sums) } else if(tb_size == 512) { - hipLaunchKernelSynchronous(dot_kernel<4, 4, T, 512>, - dim3(block_cnt), dim3(tb_size), nullptr, coherent_ev, - d_a, d_b, sums); + NT_KERNEL_LAUNCH_DOT_SYNC(dot_kernel, 4, 4, T, 512, dim3(block_cnt), dim3(tb_size), nullptr, coherent_ev, d_a, d_b, sums) } } else if(elements_per_lane == 2 && chunks_per_block == 4) { if(tb_size == 1024) { - hipLaunchKernelSynchronous(dot_kernel<2, 4, T, 1024>, - dim3(block_cnt), dim3(tb_size), nullptr, coherent_ev, - d_a, d_b, sums); + NT_KERNEL_LAUNCH_DOT_SYNC(dot_kernel, 2, 4, T, 1024, dim3(block_cnt), dim3(tb_size), nullptr, coherent_ev, d_a, d_b, sums) } else if(tb_size == 512) { - hipLaunchKernelSynchronous(dot_kernel<2, 4, T, 512>, - dim3(block_cnt), dim3(tb_size), nullptr, coherent_ev, - d_a, d_b, sums); + NT_KERNEL_LAUNCH_DOT_SYNC(dot_kernel, 2, 4, T, 512, dim3(block_cnt), dim3(tb_size), nullptr, coherent_ev, d_a, d_b, sums) } } else - hipLaunchKernelSynchronous(dot_kernel<4, 2, T, 1024>, - dim3(block_cnt), dim3(tb_size), nullptr, coherent_ev, - d_a, d_b, sums); + NT_KERNEL_LAUNCH_DOT_SYNC(dot_kernel, 4, 2, T, 1024, dim3(block_cnt), dim3(tb_size), nullptr, coherent_ev, d_a, d_b, sums) T sum{0}; for (auto i = 0u; i != block_cnt; ++i) diff --git a/babel.so/src/rvs_stress.cpp b/babel.so/src/rvs_stress.cpp index e4faf3e69..3fd2f05af 100644 --- a/babel.so/src/rvs_stress.cpp +++ b/babel.so/src/rvs_stress.cpp @@ -32,61 +32,75 @@ static bool triad_only = false; bool event_timing = false; std::string module_name{"babel"}; +// Total no. of babel subtests +const int total_babel_subtests = 7; template void check_solution(const unsigned int ntimes, std::vector& a, std::vector& b, std::vector& c, T& sum, uint64_t); template -void run_stress(std::pair device, int num_times, int ARRAY_SIZE, bool output_as_csv, bool mibibytes, int subtest, - uint16_t dwords_per_lane, uint16_t chunks_per_block, uint16_t tb_size, bool json, std::string action, int rwtest); +bool run_stress(std::pair device, int num_times, int ARRAY_SIZE, bool output_as_csv, bool mibibytes, + uint16_t dwords_per_lane, uint16_t chunks_per_block, uint16_t tb_size, bool json, std::string action, subtest *test, + const std::string& data_init, const std::string& nontemporal, uint64_t duration, bool sustained); template -void run_triad(std::pair device, int num_times, int ARRAY_SIZE, bool output_as_csv, bool mibibytes, int subtest, - uint16_t dwords_per_lane, uint16_t chunks_per_block, uint16_t tb_size, bool json, std::string action, int rwtest); +bool run_triad(std::pair device, int num_times, int ARRAY_SIZE, bool output_as_csv, bool mibibytes, + uint16_t dwords_per_lane, uint16_t chunks_per_block, uint16_t tb_size, bool json, std::string action, subtest *test, + const std::string& data_init, const std::string& nontemporal, uint64_t duration, bool sustained); void parseArguments(int argc, char *argv[]); -void run_babel(std::pair device, int num_times, int array_size, bool output_csv, bool mibibytes, int test_type, int subtest, - uint16_t dwords_per_lane, uint16_t chunks_per_block, uint16_t tb_size, bool json, std::string action, int rwtest) { +bool run_babel(std::pair device, int num_times, int array_size, bool output_csv, bool mibibytes, int test_type, + uint16_t dwords_per_lane, uint16_t chunks_per_block, uint16_t tb_size, bool json, std::string action, subtest *test, + const std::string& data_init, const std::string& nontemporal, uint64_t duration, bool sustained) { + + bool result = false; switch(test_type) { case FLOAT_TEST: - run_stress(device, num_times, array_size, output_csv, mibibytes, subtest, dwords_per_lane, chunks_per_block, tb_size, - json, action, rwtest); + result = run_stress(device, num_times, array_size, output_csv, mibibytes, dwords_per_lane, chunks_per_block, tb_size, + json, action, test, data_init, nontemporal, duration, sustained); break; case DOUBLE_TEST: - run_stress(device, num_times, array_size, output_csv, mibibytes, subtest, dwords_per_lane, chunks_per_block, tb_size, - json, action, rwtest); + result = run_stress(device, num_times, array_size, output_csv, mibibytes, dwords_per_lane, chunks_per_block, tb_size, + json, action, test, data_init, nontemporal, duration, sustained); break; case TRAID_FLOAT: - run_triad(device, num_times, array_size, output_csv, mibibytes, subtest, dwords_per_lane, chunks_per_block, tb_size, - json, action, rwtest); + result = run_triad(device, num_times, array_size, output_csv, mibibytes, dwords_per_lane, chunks_per_block, tb_size, + json, action, test, data_init, nontemporal, duration, sustained); break; case TRIAD_DOUBLE: - run_triad(device, num_times, array_size, output_csv, mibibytes, subtest, dwords_per_lane, chunks_per_block, tb_size, - json, action, rwtest); + result = run_triad(device, num_times, array_size, output_csv, mibibytes, dwords_per_lane, chunks_per_block, tb_size, + json, action, test, data_init, nontemporal, duration, sustained); break; default: std::cout << "\n specify a valid testnumber"; break; } + + return result; } template -void run_stress(std::pair device, int num_times, int ARRAY_SIZE, bool output_as_csv, bool mibibytes, int subtest, - uint16_t dwords_per_lane, uint16_t chunks_per_block, uint16_t tb_size, bool json, std::string action, int rwtest) +bool run_stress(std::pair device, int num_times, int ARRAY_SIZE, bool output_as_csv, bool mibibytes, + uint16_t dwords_per_lane, uint16_t chunks_per_block, uint16_t tb_size, bool json, std::string action, subtest *test, + const std::string& data_init, const std::string& nontemporal, uint64_t duration, bool sustained) { std::string msg; std::streamsize ss = std::cout.precision(); std::stringstream sstr; - auto desc = action_descriptor{action, module_name, device.second}; + bool time_based = (duration > 0); + if (!output_as_csv) { - msg = "Running kernels " + std::to_string(num_times) + " times, " ; + if (time_based) + msg = "Running kernels for " + std::to_string(duration) + " ms, "; + else + msg = "Running kernels " + std::to_string(num_times) + " times, " ; if (sizeof(T) == sizeof(float)) @@ -118,17 +132,6 @@ void run_stress(std::pair device, int num_times, int ARRAY_SIZE, } - //json - if (json){ - std::string scale = mibibytes ? "MiB" : "MB"; - auto arr_size = mibibytes ? ARRAY_SIZE*sizeof(T)*pow(2.0, -20.0) : - ARRAY_SIZE*sizeof(T)*1.0E-6; - auto total_size = mibibytes ? 3.0*ARRAY_SIZE*sizeof(T)*pow(2.0, -20.0) : - 3.0*ARRAY_SIZE*sizeof(T)*1.0E-6; - log_to_json(desc, rvs::logresults,"Array size", std::to_string(arr_size), - "Total size", std::to_string(total_size), - "Iterations", std::to_string(num_times) ); - } // Create host vectors std::vector a(ARRAY_SIZE); @@ -138,73 +141,275 @@ void run_stress(std::pair device, int num_times, int ARRAY_SIZE, // Result of the Dot kernel T sum; - Stream *stream; - // Use the HIP implementation - stream = new HIPStream(ARRAY_SIZE, event_timing, device.first, dwords_per_lane, chunks_per_block, tb_size); + HIPStream *stream = new HIPStream(ARRAY_SIZE, event_timing, device.first, dwords_per_lane, chunks_per_block, tb_size, nontemporal); + + if (data_init == "gpu_norm_dist") { + stream->init_arrays_normdist(static_cast(0.0), static_cast(1.0), true, a, b, c); + } else if (data_init == "cpu_norm_dist") { + stream->init_arrays_normdist(static_cast(0.0), static_cast(1.0), false, a, b, c); + } else if (data_init == "zero_init") { + stream->init_arrays(T{0}, T{0}, T{0}); + } else { + stream->init_arrays(startA, startB, startC); + } - stream->init_arrays(startA, startB, startC); + // shared by both sustained and normal paths + std::string labels[total_babel_subtests] = {"Read","Write","Copy","Mul","Add","Triad","Dot"}; + size_t sizes[total_babel_subtests] = { + 1 * sizeof(T) * ARRAY_SIZE, + 1 * sizeof(T) * ARRAY_SIZE, + 2 * sizeof(T) * ARRAY_SIZE, + 2 * sizeof(T) * ARRAY_SIZE, + 3 * sizeof(T) * ARRAY_SIZE, + 3 * sizeof(T) * ARRAY_SIZE, + 2 * sizeof(T) * ARRAY_SIZE + }; + bool test_enable[total_babel_subtests] = { + test->read, test->write, test->copy, test->mul, test->add, test->triad, test->dot + }; + const double bw_scale = mibibytes ? pow(2.0, -20.0) : 1.0E-6; + uint64_t arr_size = (uint64_t)ARRAY_SIZE * sizeof(T); + uint64_t total_size = 3ULL * ARRAY_SIZE * sizeof(T); + auto format_bw = [](double val) -> std::string { + std::ostringstream oss; + oss << std::fixed << std::setprecision(3) << val; + return oss.str(); + }; + std::string iter_key = time_based ? "duration_ms" : "iterations"; + std::string iter_val = time_based ? std::to_string(duration) : std::to_string(num_times); + + // ---- Sustained mode: back-to-back launches, single sync per kernel ---- + if (sustained) + { + stream->set_sustained_mode(true); + + std::chrono::high_resolution_clock::time_point t1, t2; + + if (output_as_csv) + { + sstr << "gpu_id" << csv_separator + << "function" << csv_separator + << "num_times" << csv_separator + << "n_elements" << csv_separator + << "sizeof" << csv_separator + << ((mibibytes) ? "max_mibytes_per_sec" : "max_mbytes_per_sec") << csv_separator + << ((mibibytes) ? "mibps_at_min_t" : "mbps_at_min_t") << csv_separator + << ((mibibytes) ? "mibps_at_max_t" : "mbps_at_max_t") << csv_separator + << ((mibibytes) ? "mibps_at_avg_t" : "mbps_at_avg_t") << std::endl; + } + else + { + sstr << "\n---------------------------------------------------------------------------------" << std::endl + << std::left << std::setw(12) << "GPU Id" + << std::left << std::setw(12) << "Function" + << std::left << std::setw(15) << ((mibibytes) ? "MiBytes/sec" : "MBytes/sec") + << std::left << std::setw(15) << ((mibibytes) ? "Max MiB/s" : "Max MB/s") + << std::left << std::setw(15) << ((mibibytes) ? "Min MiB/s" : "Min MB/s") + << std::left << std::setw(15) << ((mibibytes) ? "Avg MiB/s" : "Avg MB/s") + << std::endl + << "---------------------------------------------------------------------------------" << std::endl + << std::fixed; + } + + void* json_root = nullptr; + void* json_results = nullptr; + if (json) { + unsigned int sec, usec; + rvs::lp::get_ticks(&sec, &usec); + json_root = rvs::lp::LogRecordCreate(module_name.c_str(), action.c_str(), + rvs::logresults, sec, usec, true); + if (json_root) { + rvs::lp::AddString(json_root, "gpu_id", std::to_string(device.second)); + uint16_t gpu_index = 0; + rvs::gpulist::gpu2gpuindex(device.second, &gpu_index); + rvs::lp::AddString(json_root, "gpu_index", std::to_string(gpu_index)); + rvs::lp::AddString(json_root, "array_size", std::to_string(arr_size)); + rvs::lp::AddString(json_root, "total_size", std::to_string(total_size)); + rvs::lp::AddString(json_root, iter_key, iter_val); + json_results = rvs::lp::JsonNestedListCreate("results", rvs::logresults); + } + } + + for (int i = 0; i < total_babel_subtests; ++i) + { + if (!test_enable[i]) continue; + + uint64_t actual_k = 0; + t1 = std::chrono::high_resolution_clock::now(); + auto duration_limit = std::chrono::milliseconds(duration); + for (uint64_t k = 0; !time_based ? (k < (uint64_t)num_times) : true; k++) { + if (time_based) { + if (std::chrono::high_resolution_clock::now() - t1 >= duration_limit) break; + } + switch (i) { + case 0: stream->read(); break; + case 1: stream->write(); break; + case 2: stream->copy(); break; + case 3: stream->mul(); break; + case 4: stream->add(); break; + case 5: stream->triad(); break; + case 6: stream->dot(); break; + } + actual_k++; + } + // dot() always syncs internally; for all others, issue the single batch sync + if (i != 6) stream->sustained_sync(); + t2 = std::chrono::high_resolution_clock::now(); + + double total_s = std::chrono::duration_cast>(t2 - t1).count(); + double avg_s = total_s / actual_k; + double avg_bw = (bw_scale * sizes[i]) / avg_s; + + if (output_as_csv) + { + sstr << device.second << csv_separator + << labels[i] << csv_separator + << actual_k << csv_separator + << ARRAY_SIZE << csv_separator + << sizeof(T) << csv_separator + << avg_bw << csv_separator + << avg_bw << csv_separator + << avg_bw << csv_separator + << avg_bw << std::endl; + } + else + { + sstr << std::left << std::setw(12) << device.second + << std::left << std::setw(12) << labels[i] + << std::left << std::setw(15) << format_bw(avg_bw) + << std::left << std::setw(15) << format_bw(avg_bw) + << std::left << std::setw(15) << format_bw(avg_bw) + << std::left << std::setw(15) << format_bw(avg_bw) + << std::endl; + } + + if (json && json_results) { + const char* key = mibibytes ? "mibytes_per_sec" : "mbytes_per_sec"; + const char* peak_key = mibibytes ? "max_mibytes_per_sec" : "max_mbytes_per_sec"; + const char* worst_key = mibibytes ? "min_mibytes_per_sec" : "min_mbytes_per_sec"; + const char* avg_key = mibibytes ? "avg_mibytes_per_sec" : "avg_mbytes_per_sec"; + void* subtest_node = rvs::lp::LogRecordCreate(module_name.c_str(), + labels[i].c_str(), rvs::logresults, 0, 0, true); + if (subtest_node) { + rvs::lp::AddString(subtest_node, "subtest", labels[i]); + rvs::lp::AddString(subtest_node, key, format_bw(avg_bw)); + rvs::lp::AddString(subtest_node, peak_key, format_bw(avg_bw)); + rvs::lp::AddString(subtest_node, worst_key, format_bw(avg_bw)); + rvs::lp::AddString(subtest_node, avg_key, format_bw(avg_bw)); + rvs::lp::AddString(subtest_node, "pass", "true"); + rvs::lp::AddNode(json_results, subtest_node); + } + } + } + + sstr << "---------------------------------------------------------------------------------" << std::endl; + rvs::lp::Log(sstr.str(), rvs::logresults); + + if (json && json_root && json_results) { + rvs::lp::AddNode(json_root, json_results); + rvs::lp::LogRecordFlush(json_root, true); + } + + stream->set_sustained_mode(false); + delete stream; + return true; + } + // ---- End sustained mode ---- // List of times - std::vector> rwtimings(2); - std::vector> timings(5); + std::vector> timings(total_babel_subtests); // Declare timers std::chrono::high_resolution_clock::time_point t1, t2; - for (unsigned int k = 0; k < num_times; k++) + auto loop_start = std::chrono::high_resolution_clock::now(); + auto duration_limit = std::chrono::milliseconds(duration); + uint64_t actual_iterations = 0; + + // Main loop - run each babel subtest if enabled + // When duration > 0: run until elapsed time exceeds duration + // When duration == 0: run for num_times iterations + for (uint64_t k = 0; !time_based ? (k < (uint64_t)num_times) : true; k++) { - // Execute Read - t1 = std::chrono::high_resolution_clock::now(); - stream->read(); - t2 = std::chrono::high_resolution_clock::now(); - rwtimings[0].push_back(std::chrono::duration_cast >(t2 - t1).count()); + if (time_based) { + auto elapsed = std::chrono::high_resolution_clock::now() - loop_start; + if (elapsed >= duration_limit) + break; + } - // Execute Write - t1 = std::chrono::high_resolution_clock::now(); - stream->write(); - t2 = std::chrono::high_resolution_clock::now(); - rwtimings[1].push_back(std::chrono::duration_cast >(t2 - t1).count()); - } + if(test->read) { + // Execute Read + t1 = std::chrono::high_resolution_clock::now(); + stream->read(); + t2 = std::chrono::high_resolution_clock::now(); + timings[0].push_back(std::chrono::duration_cast >(t2 - t1).count()); + } - // Main loop - for (unsigned int k = 0; k < num_times; k++) - { - // Execute Copy - t1 = std::chrono::high_resolution_clock::now(); - stream->copy(); - t2 = std::chrono::high_resolution_clock::now(); - timings[0].push_back(std::chrono::duration_cast >(t2 - t1).count()); + if(test->write) { + // Execute Write + t1 = std::chrono::high_resolution_clock::now(); + stream->write(); + t2 = std::chrono::high_resolution_clock::now(); + timings[1].push_back(std::chrono::duration_cast >(t2 - t1).count()); + } - // Execute Mul - t1 = std::chrono::high_resolution_clock::now(); - stream->mul(); - t2 = std::chrono::high_resolution_clock::now(); - timings[1].push_back(std::chrono::duration_cast >(t2 - t1).count()); + if(test->copy) { + // Execute Copy + t1 = std::chrono::high_resolution_clock::now(); + stream->copy(); + t2 = std::chrono::high_resolution_clock::now(); + timings[2].push_back(std::chrono::duration_cast >(t2 - t1).count()); + } - // Execute Add - t1 = std::chrono::high_resolution_clock::now(); - stream->add(); - t2 = std::chrono::high_resolution_clock::now(); - timings[2].push_back(std::chrono::duration_cast >(t2 - t1).count()); + if(test->mul) { + // Execute Mul + t1 = std::chrono::high_resolution_clock::now(); + stream->mul(); + t2 = std::chrono::high_resolution_clock::now(); + timings[3].push_back(std::chrono::duration_cast >(t2 - t1).count()); + } - // Execute Triad - t1 = std::chrono::high_resolution_clock::now(); - stream->triad(); - t2 = std::chrono::high_resolution_clock::now(); - timings[3].push_back(std::chrono::duration_cast >(t2 - t1).count()); + if(test->add) { + // Execute Add + t1 = std::chrono::high_resolution_clock::now(); + stream->add(); + t2 = std::chrono::high_resolution_clock::now(); + timings[4].push_back(std::chrono::duration_cast >(t2 - t1).count()); + } - // Execute Dot - t1 = std::chrono::high_resolution_clock::now(); - sum = stream->dot(); - t2 = std::chrono::high_resolution_clock::now(); - timings[4].push_back(std::chrono::duration_cast >(t2 - t1).count()); + if(test->triad) { + // Execute Triad + t1 = std::chrono::high_resolution_clock::now(); + stream->triad(); + t2 = std::chrono::high_resolution_clock::now(); + timings[5].push_back(std::chrono::duration_cast >(t2 - t1).count()); + } + + if(test->dot) { + // Execute Dot + t1 = std::chrono::high_resolution_clock::now(); + sum = stream->dot(); + t2 = std::chrono::high_resolution_clock::now(); + timings[6].push_back(std::chrono::duration_cast >(t2 - t1).count()); + } + + actual_iterations++; + } + + uint64_t effective_num_times = time_based ? actual_iterations : (uint64_t)num_times; + if (time_based) { + auto total_elapsed = std::chrono::duration_cast>( + std::chrono::high_resolution_clock::now() - loop_start).count(); + msg = "Completed " + std::to_string(actual_iterations) + " iterations in " + + std::to_string(total_elapsed) + " seconds"; + rvs::lp::Log(msg, rvs::logresults); } // Check solutions stream->read_arrays(a, b, c); - check_solution(num_times, a, b, c, sum, ARRAY_SIZE); +//check_solution(num_times, a, b, c, sum, ARRAY_SIZE); sstr.str( std::string() ); sstr.clear(); if (output_as_csv) @@ -215,148 +420,137 @@ void run_stress(std::pair device, int num_times, int ARRAY_SIZE, << "n_elements" << csv_separator << "sizeof" << csv_separator << ((mibibytes) ? "max_mibytes_per_sec" : "max_mbytes_per_sec") << csv_separator - << "min_runtime" << csv_separator - << "max_runtime" << csv_separator - << "avg_runtime" << std::endl; + << ((mibibytes) ? "mibps_at_min_t" : "mbps_at_min_t") << csv_separator + << ((mibibytes) ? "mibps_at_max_t" : "mbps_at_max_t") << csv_separator + << ((mibibytes) ? "mibps_at_avg_t" : "mbps_at_avg_t") << std::endl; } else { - sstr << "\n------------------------------------------------------------------------" << std::endl + sstr << "\n---------------------------------------------------------------------------------" << std::endl << std::left << std::setw(12) << "GPU Id" << std::left << std::setw(12) << "Function" - << std::left << std::setw(12) << ((mibibytes) ? "MiBytes/sec" : "MBytes/sec") - << std::left << std::setw(12) << "Min (sec)" - << std::left << std::setw(12) << "Max" - << std::left << std::setw(12) << "Average" + << std::left << std::setw(15) << ((mibibytes) ? "MiBytes/sec" : "MBytes/sec") + << std::left << std::setw(15) << ((mibibytes) ? "Max MiB/s" : "Max MB/s") + << std::left << std::setw(15) << ((mibibytes) ? "Min MiB/s" : "Min MB/s") + << std::left << std::setw(15) << ((mibibytes) ? "Avg MiB/s" : "Avg MB/s") << std::endl - << "------------------------------------------------------------------------" << std::endl + << "---------------------------------------------------------------------------------" << std::endl << std::fixed; } - std::string rwlabels[2] = {"Read", "Write"}; - size_t rwsizes[2] = { - sizeof(T) * ARRAY_SIZE, - sizeof(T) * ARRAY_SIZE - }; - - for (int i = 0; i < rwtest; i++) - { - // Get min/max; ignore the first result - auto minmax = std::minmax_element(rwtimings[i].begin()+1, rwtimings[i].end()); - - // Calculate average; ignore the first result - double average = std::accumulate(rwtimings[i].begin()+1, rwtimings[i].end(), 0.0) / (double)(num_times - 1); - // Display results - if (output_as_csv) - { - sstr - << device.second << csv_separator - << rwlabels[i] << csv_separator - << num_times << csv_separator - << ARRAY_SIZE << csv_separator - << sizeof(T) << csv_separator - << ((mibibytes) ? pow(2.0, -20.0) : 1.0E-6) * rwsizes[i] / (*minmax.first) << csv_separator - << *minmax.first << csv_separator - << *minmax.second << csv_separator - << average - << std::endl; - } - else - { - sstr - << std::left << std::setw(12) << device.second - << std::left << std::setw(12) << rwlabels[i] - << std::left << std::setw(12) << std::setprecision(3) << - ((mibibytes) ? pow(2.0, -20.0) : 1.0E-6) * rwsizes[i] / (*minmax.first) - << std::left << std::setw(12) << std::setprecision(5) << *minmax.first - << std::left << std::setw(12) << std::setprecision(5) << *minmax.second - << std::left << std::setw(12) << std::setprecision(5) << average - << std::endl; - } - if (json){ - log_to_json(desc, rvs::logresults, "Function",std::string(rwlabels[i]), - "MBytes/sec", (mibibytes) ? - std::to_string(pow(2.0, -20.0)) : std::to_string((1.0E-6) * rwsizes[i] / (*minmax.first)), - "Min(s)",std::to_string( *minmax.first), - "Max(s)", std::to_string(*minmax.second), - "Average(s)", std::to_string(average), - "pass", "true"); + // Build single JSON node per GPU: metadata + nested "results" array + void* json_root = nullptr; + void* json_results = nullptr; + if (json) { + unsigned int sec, usec; + rvs::lp::get_ticks(&sec, &usec); + json_root = rvs::lp::LogRecordCreate(module_name.c_str(), action.c_str(), + rvs::logresults, sec, usec, true); + if (json_root) { + rvs::lp::AddString(json_root, "gpu_id", std::to_string(device.second)); + uint16_t gpu_index = 0; + rvs::gpulist::gpu2gpuindex(device.second, &gpu_index); + rvs::lp::AddString(json_root, "gpu_index", std::to_string(gpu_index)); + rvs::lp::AddString(json_root, "array_size", std::to_string(arr_size)); + rvs::lp::AddString(json_root, "total_size", std::to_string(total_size)); + rvs::lp::AddString(json_root, iter_key, iter_val); + json_results = rvs::lp::JsonNestedListCreate("results", rvs::logresults); } } - //rvs::lp::Log(sstr.str(), rvs::logresults); - std::string labels[5] = {"Copy", "Mul", "Add", "Triad", "Dot"}; - size_t sizes[5] = { - 2 * sizeof(T) * ARRAY_SIZE, - 2 * sizeof(T) * ARRAY_SIZE, - 3 * sizeof(T) * ARRAY_SIZE, - 3 * sizeof(T) * ARRAY_SIZE, - 2 * sizeof(T) * ARRAY_SIZE - }; - - for (int i = 0; i < subtest; i++) + // Display babel subtest results + for (int i = 0; i < total_babel_subtests; i++) { - // Get min/max; ignore the first result - auto minmax = std::minmax_element(timings[i].begin()+1, timings[i].end()); - //sstr.str( std::string() ); - //sstr.clear(); - // Calculate average; ignore the first result - double average = std::accumulate(timings[i].begin()+1, timings[i].end(), 0.0) / (double)(num_times - 1); - // Display results - if (output_as_csv) - { - sstr - << device.second << csv_separator - << labels[i] << csv_separator - << num_times << csv_separator - << ARRAY_SIZE << csv_separator - << sizeof(T) << csv_separator - << ((mibibytes) ? pow(2.0, -20.0) : 1.0E-6) * sizes[i] / (*minmax.first) << csv_separator - << *minmax.first << csv_separator - << *minmax.second << csv_separator - << average - << std::endl; + if(test_enable[i]) { + + // Get min/max; ignore the first result + auto minmax = std::minmax_element(timings[i].begin()+1, timings[i].end()); + + // Calculate average; ignore the first result + double average = std::accumulate(timings[i].begin()+1, timings[i].end(), 0.0) / (double)(effective_num_times - 1); + // Display results + if (output_as_csv) + { + sstr + << device.second << csv_separator + << labels[i] << csv_separator + << effective_num_times << csv_separator + << ARRAY_SIZE << csv_separator + << sizeof(T) << csv_separator + << bw_scale * sizes[i] / (*minmax.first) << csv_separator + << bw_scale * sizes[i] / (*minmax.first) << csv_separator + << bw_scale * sizes[i] / (*minmax.second) << csv_separator + << bw_scale * sizes[i] / average + << std::endl; + } + else + { + sstr + << std::left << std::setw(12) << device.second + << std::left << std::setw(12) << labels[i] + << std::left << std::setw(15) << format_bw(bw_scale * sizes[i] / (*minmax.first)) + << std::left << std::setw(15) << format_bw(bw_scale * sizes[i] / (*minmax.first)) + << std::left << std::setw(15) << format_bw(bw_scale * sizes[i] / (*minmax.second)) + << std::left << std::setw(15) << format_bw(bw_scale * sizes[i] / average) + << std::endl; + } + if (json && json_results) { + const char *key = mibibytes ? "mibytes_per_sec" : "mbytes_per_sec"; + const char *peak_key = mibibytes ? "max_mibytes_per_sec" : "max_mbytes_per_sec"; + const char *worst_key = mibibytes ? "min_mibytes_per_sec" : "min_mbytes_per_sec"; + const char *avg_key = mibibytes ? "avg_mibytes_per_sec" : "avg_mbytes_per_sec"; + void* subtest_node = rvs::lp::LogRecordCreate(module_name.c_str(), + labels[i].c_str(), rvs::logresults, 0, 0, true); + if (subtest_node) { + rvs::lp::AddString(subtest_node, "subtest", labels[i]); + rvs::lp::AddString(subtest_node, key, + format_bw(bw_scale * sizes[i] / (*minmax.first))); + rvs::lp::AddString(subtest_node, peak_key, + format_bw(bw_scale * sizes[i] / (*minmax.first))); + rvs::lp::AddString(subtest_node, worst_key, + format_bw(bw_scale * sizes[i] / (*minmax.second))); + rvs::lp::AddString(subtest_node, avg_key, + format_bw(bw_scale * sizes[i] / average)); + rvs::lp::AddString(subtest_node, "pass", "true"); + rvs::lp::AddNode(json_results, subtest_node); + } + } } - else - { - sstr - << std::left << std::setw(12) << device.second - << std::left << std::setw(12) << labels[i] - << std::left << std::setw(12) << std::setprecision(3) << - ((mibibytes) ? pow(2.0, -20.0) : 1.0E-6) * sizes[i] / (*minmax.first) - << std::left << std::setw(12) << std::setprecision(5) << *minmax.first - << std::left << std::setw(12) << std::setprecision(5) << *minmax.second - << std::left << std::setw(12) << std::setprecision(5) << average - << std::endl; - } - if (json){ - log_to_json(desc, rvs::logresults, "Function",std::string(labels[i]), - "MBytes/sec", (mibibytes) ? - std::to_string(pow(2.0, -20.0)) : std::to_string((1.0E-6) * sizes[i] / (*minmax.first)), - "Min(s)",std::to_string( *minmax.first), - "Max(s)", std::to_string(*minmax.second), - "Average(s)", std::to_string(average), - "pass", "true"); - } } + sstr - << "------------------------------------------------------------------------" << std::endl; + << "---------------------------------------------------------------------------------" << std::endl; rvs::lp::Log(sstr.str(), rvs::logresults); + + if (json && json_root && json_results) { + rvs::lp::AddNode(json_root, json_results); + rvs::lp::LogRecordFlush(json_root, true); + } + delete stream; + return true; } template -void run_triad(std::pair device, int num_times, int ARRAY_SIZE, bool output_as_csv, bool mibibytes, int subtest, - uint16_t dwords_per_lane, uint16_t chunks_per_block, uint16_t tb_size, bool json, std::string action, int rwtest) +bool run_triad(std::pair device, int num_times, int ARRAY_SIZE, bool output_as_csv, bool mibibytes, + uint16_t dwords_per_lane, uint16_t chunks_per_block, uint16_t tb_size, bool json, std::string action, subtest *test, + const std::string& data_init, const std::string& nontemporal, uint64_t duration, bool sustained) { std::string msg; - auto desc = action_descriptor{action, module_name, device.second}; triad_only = true; + bool time_based = (duration > 0); std::stringstream sstr; + uint64_t arr_size = (uint64_t)ARRAY_SIZE * sizeof(T); + uint64_t total_size = 3ULL * ARRAY_SIZE * sizeof(T); + std::string iter_key = time_based ? "duration_ms" : "iterations"; + std::string iter_val = time_based ? std::to_string(duration) : std::to_string(num_times); if (!output_as_csv) { - msg = "Running triad " + std::to_string (num_times) + " times,"; + if (time_based) + msg = "Running triad for " + std::to_string(duration) + " ms,"; + else + msg = "Running triad " + std::to_string (num_times) + " times,"; msg += "Number of elements: " + std::to_string(ARRAY_SIZE) + ", "; if (sizeof(T) == sizeof(float)) @@ -383,16 +577,6 @@ void run_triad(std::pair device, int num_times, int ARRAY_SIZE, b << " (=" << 3.0*ARRAY_SIZE*sizeof(T)*1.0E-6 << " MB)" << std::endl; } rvs::lp::Log(sstr.str(), rvs::logresults); - if (json){ - std::string scale = mibibytes ? "MiB" : "MB"; - auto arr_size = mibibytes ? ARRAY_SIZE*sizeof(T)*pow(2.0, -20.0) : - ARRAY_SIZE*sizeof(T)*1.0E-6; - auto total_size = mibibytes ? 3.0*ARRAY_SIZE*sizeof(T)*pow(2.0, -20.0) : - 3.0*ARRAY_SIZE*sizeof(T)*1.0E-6; - log_to_json(desc, rvs::logresults,"Array size", std::to_string(arr_size), - "Total size", std::to_string(total_size), - "Iterations", std::to_string(num_times) ); - } std::cout.precision(ss); } sstr.str( std::string() ); @@ -402,33 +586,142 @@ void run_triad(std::pair device, int num_times, int ARRAY_SIZE, b std::vector b(ARRAY_SIZE); std::vector c(ARRAY_SIZE); - Stream *stream; - // Use the HIP implementation - stream = new HIPStream(ARRAY_SIZE, event_timing, device.first, dwords_per_lane, chunks_per_block, tb_size); + HIPStream *stream = new HIPStream(ARRAY_SIZE, event_timing, device.first, dwords_per_lane, chunks_per_block, tb_size, nontemporal); + + if (data_init == "gpu_norm_dist") { + stream->init_arrays_normdist(static_cast(0.0), static_cast(1.0), true, a, b, c); + } else if (data_init == "cpu_norm_dist") { + stream->init_arrays_normdist(static_cast(0.0), static_cast(1.0), false, a, b, c); + } else if (data_init == "zero_init") { + stream->init_arrays(T{0}, T{0}, T{0}); + } else { + stream->init_arrays(startA, startB, startC); + } + + // ---- Sustained mode: back-to-back triad launches, single sync ---- + if (sustained) + { + stream->set_sustained_mode(true); + + std::chrono::high_resolution_clock::time_point t1, t2; + uint64_t actual_k = 0; + t1 = std::chrono::high_resolution_clock::now(); + auto duration_limit = std::chrono::milliseconds(duration); + for (uint64_t k = 0; !time_based ? (k < (uint64_t)num_times) : true; k++) { + if (time_based) { + if (std::chrono::high_resolution_clock::now() - t1 >= duration_limit) break; + } + stream->triad(); + actual_k++; + } + stream->sustained_sync(); + t2 = std::chrono::high_resolution_clock::now(); + + double runtime = std::chrono::duration_cast>(t2 - t1).count(); + double total_bytes = 3.0 * sizeof(T) * ARRAY_SIZE * actual_k; + double bandwidth = ((mibibytes) ? pow(2.0, -30.0) : 1.0E-9) * (total_bytes / runtime); + + std::stringstream sstr2; + if (output_as_csv) + { + sstr2 << "gpu_id" << csv_separator + << "function" << csv_separator + << "num_times" << csv_separator + << "n_elements" << csv_separator + << "sizeof" << csv_separator + << ((mibibytes) ? "gibytes_per_sec" : "gbytes_per_sec") << csv_separator + << "runtime" + << std::endl + << device.second << csv_separator + << "Triad" << csv_separator + << actual_k << csv_separator + << ARRAY_SIZE << csv_separator + << sizeof(T) << csv_separator + << bandwidth << csv_separator + << runtime + << std::endl; + } + else + { + sstr2 << "--------------------------------" + << std::endl << std::fixed + << "GPU Id: " << std::left << device.second << std::endl + << "Runtime (seconds): " << std::left << std::setprecision(5) + << runtime << std::endl + << "Bandwidth (" << ((mibibytes) ? "GiB/s" : "GB/s") << "): " + << std::left << std::setprecision(3) + << bandwidth << std::endl; + } + rvs::lp::Log(sstr2.str(), rvs::logresults); + + if (json) { + unsigned int sec, usec; + rvs::lp::get_ticks(&sec, &usec); + void* json_root = rvs::lp::LogRecordCreate(module_name.c_str(), action.c_str(), + rvs::logresults, sec, usec, true); + if (json_root) { + rvs::lp::AddString(json_root, "gpu_id", std::to_string(device.second)); + uint16_t gpu_index = 0; + rvs::gpulist::gpu2gpuindex(device.second, &gpu_index); + rvs::lp::AddString(json_root, "gpu_index", std::to_string(gpu_index)); + rvs::lp::AddString(json_root, "array_size", std::to_string(arr_size)); + rvs::lp::AddString(json_root, "total_size", std::to_string(total_size)); + rvs::lp::AddString(json_root, iter_key, iter_val); + const char* bw_key = mibibytes ? "bandwidth_gib_per_sec" : "bandwidth_gb_per_sec"; + rvs::lp::AddString(json_root, "runtime_sec", std::to_string(runtime)); + rvs::lp::AddString(json_root, bw_key, std::to_string(bandwidth)); + rvs::lp::AddString(json_root, "pass", "true"); + rvs::lp::LogRecordFlush(json_root, true); + } + } - stream->init_arrays(startA, startB, startC); + stream->set_sustained_mode(false); + delete stream; + return true; + } + // ---- End sustained mode ---- // Declare timers std::chrono::high_resolution_clock::time_point t1, t2; + uint64_t actual_iterations = 0; + // Run triad in loop t1 = std::chrono::high_resolution_clock::now(); - for (unsigned int k = 0; k < num_times; k++) - { - stream->triad(); + if (time_based) { + auto duration_limit = std::chrono::milliseconds(duration); + while (true) { + auto elapsed = std::chrono::high_resolution_clock::now() - t1; + if (elapsed >= duration_limit) + break; + stream->triad(); + actual_iterations++; + } + } else { + for (unsigned int k = 0; k < num_times; k++) + { + stream->triad(); + } + actual_iterations = num_times; } t2 = std::chrono::high_resolution_clock::now(); double runtime = std::chrono::duration_cast >(t2 - t1).count(); + if (time_based) { + msg = "Completed " + std::to_string(actual_iterations) + " triad iterations in " + + std::to_string(runtime) + " seconds"; + rvs::lp::Log(msg, rvs::logresults); + } + // Check solutions T sum = 0.0; stream->read_arrays(a, b, c); - check_solution(num_times, a, b, c, sum, ARRAY_SIZE); +// check_solution(num_times, a, b, c, sum, ARRAY_SIZE); // Display timing results - double total_bytes = 3 * sizeof(T) * ARRAY_SIZE * num_times; + double total_bytes = 3 * sizeof(T) * ARRAY_SIZE * actual_iterations; double bandwidth = ((mibibytes) ? pow(2.0, -30.0) : 1.0E-9) * (total_bytes / runtime); if (output_as_csv) @@ -444,7 +737,7 @@ void run_triad(std::pair device, int num_times, int ARRAY_SIZE, b << std::endl << device.second << csv_separator << "Triad" << csv_separator - << num_times << csv_separator + << actual_iterations << csv_separator << ARRAY_SIZE << csv_separator << sizeof(T) << csv_separator << bandwidth << csv_separator @@ -464,17 +757,29 @@ void run_triad(std::pair device, int num_times, int ARRAY_SIZE, b << bandwidth << std::endl; } rvs::lp::Log(sstr.str(), rvs::logresults); - if (json){ - std::string bw_field{"Bandwidth ("}; - bw_field +=(mibibytes) ? "GiB/s" : "GB/s"; - bw_field += ")"; - log_to_json(desc, rvs::logresults, - "GPU Id", std::to_string(device.second), - "Runtime (seconds)", std::to_string(runtime), - bw_field, std::to_string(bandwidth), - "pass", "true"); + if (json) { + unsigned int sec, usec; + rvs::lp::get_ticks(&sec, &usec); + void* json_root = rvs::lp::LogRecordCreate(module_name.c_str(), action.c_str(), + rvs::logresults, sec, usec, true); + if (json_root) { + rvs::lp::AddString(json_root, "gpu_id", std::to_string(device.second)); + uint16_t gpu_index = 0; + rvs::gpulist::gpu2gpuindex(device.second, &gpu_index); + rvs::lp::AddString(json_root, "gpu_index", std::to_string(gpu_index)); + rvs::lp::AddString(json_root, "array_size", std::to_string(arr_size)); + rvs::lp::AddString(json_root, "total_size", std::to_string(total_size)); + rvs::lp::AddString(json_root, iter_key, iter_val); + const char *bw_key = mibibytes ? "bandwidth_gib_per_sec" : "bandwidth_gb_per_sec"; + rvs::lp::AddString(json_root, "runtime_sec", std::to_string(runtime)); + rvs::lp::AddString(json_root, bw_key, std::to_string(bandwidth)); + rvs::lp::AddString(json_root, "pass", "true"); + rvs::lp::LogRecordFlush(json_root, true); + } } delete stream; + + return true; } template @@ -530,4 +835,3 @@ void check_solution(const unsigned int ntimes, std::vector& a, std::vector rvs::lp::Log(sstr.str() ,rvs::logerror); } } - diff --git a/build_packages_local.sh b/build_packages_local.sh new file mode 100644 index 000000000..e00a35f66 --- /dev/null +++ b/build_packages_local.sh @@ -0,0 +1,1320 @@ +#!/bin/bash +################################################################################ +# Local Build Script for Testing Package Generation +# This script mimics the GitHub Actions workflow for local testing +################################################################################ + +set -e # Exit on error + +# Configuration +GPU_FAMILY="${GPU_FAMILY:-gfx110X-all}" +BUILD_TYPE="${BUILD_TYPE:-Release}" +BUILD_TRANSFERBENCH_CLI="${BUILD_TRANSFERBENCH_CLI:-OFF}" +# TransferBench offload archs (independent of GPU_FAMILY SDK tarball selection). +# Source of truth: TransferBench build_packages_local.sh DEFAULT_GPU_TARGETS. +DEFAULT_GPU_TARGETS="gfx906;gfx908;gfx90a;gfx942;gfx950;gfx1030;gfx1100;gfx1101;gfx1102;gfx1150;gfx1151;gfx1200;gfx1201;gfx1250" +TRANSFERBENCH_GPU_TARGETS="${TRANSFERBENCH_GPU_TARGETS:-${GPU_TARGETS:-$DEFAULT_GPU_TARGETS}}" +BUILD_DIR="./build" +ROCM_INSTALL_DIR="$HOME/rocm-sdk" + +# SDK Source Configuration +# Defaults point at AMD-hosted listings (override via env or GitHub vars). +# - Nightly index: https://nightly.repo.amd.com/rocm/core/tarball/ +# - Release listing: https://repo.amd.com/rocm/tarball/ +# +# Local builds: if ROCM_VERSION is unset (channel auto, no release listing env), latest *nightly* is fetched. +# If ROCM_VERSION is set, tarball base follows format: +# - Nightly: x.y.za (e.g. 7.11.0a20260121) → nightly base URL +# - Release: X.Y.Z (e.g. 7.11.0) → release base URL +# +# ROCM_SDK_CHANNEL: nightly | release | auto (default auto). +# - nightly: latest SDK from nightly listing (ROCM_SDK_INDEX_URL). +# - release: latest X.Y.Z from release listing (ROCM_SDK_RELEASE_URL). +# - auto: release listing URL set → release discovery; else nightly index. +# +# Optional: ROCM_SDK_NIGHTLY_BASE_URL, ROCM_SDK_NIGHTLY_INDEX_URL, ROCM_SDK_RELEASE_URL (listing), +# ROCM_SDK_RELEASE_BASE_URL (tarball directory for X.Y.Z downloads). +_ROCM_NIGHTLY_INDEX_DEFAULT="https://nightly.repo.amd.com/rocm/core/tarball/" +_ROCM_NIGHTLY_BASE_DEFAULT="https://nightly.repo.amd.com/rocm/core/tarball" +_ROCM_RELEASE_LIST_DEFAULT="https://repo.amd.com/rocm/tarball/" +_ROCM_RELEASE_BASE_DEFAULT="https://repo.amd.com/rocm/tarball" + +ROCM_SDK_CHANNEL="${ROCM_SDK_CHANNEL:-auto}" +ROCM_SDK_RELEASE_URL="${ROCM_SDK_RELEASE_URL:-}" + +if [ "$ROCM_SDK_CHANNEL" = "nightly" ]; then + ROCM_SDK_RELEASE_URL="" + ROCM_SDK_BASE_URL="${ROCM_SDK_NIGHTLY_BASE_URL:-${ROCM_SDK_BASE_URL:-$_ROCM_NIGHTLY_BASE_DEFAULT}}" + ROCM_SDK_INDEX_URL="${ROCM_SDK_NIGHTLY_INDEX_URL:-${ROCM_SDK_INDEX_URL:-$_ROCM_NIGHTLY_INDEX_DEFAULT}}" +elif [ "$ROCM_SDK_CHANNEL" = "release" ]; then + ROCM_SDK_RELEASE_URL="${ROCM_SDK_RELEASE_URL:-$_ROCM_RELEASE_LIST_DEFAULT}" + if [ -z "${ROCM_SDK_BASE_URL:-}" ]; then + if [ -n "${ROCM_SDK_RELEASE_BASE_URL:-}" ]; then + ROCM_SDK_BASE_URL="${ROCM_SDK_RELEASE_BASE_URL}" + else + # Same directory as listing: strip trailing / only (not %/* — that drops /tarball when URL has no trailing slash). + ROCM_SDK_BASE_URL="${ROCM_SDK_RELEASE_URL%/}" + fi + fi + ROCM_SDK_INDEX_URL="${ROCM_SDK_INDEX_URL:-}" +else + # auto (local: no release URL → nightly latest; CI sets channel explicitly) + if [ -n "${ROCM_SDK_RELEASE_URL}" ]; then + if [ -z "${ROCM_SDK_BASE_URL:-}" ]; then + if [ -n "${ROCM_SDK_RELEASE_BASE_URL:-}" ]; then + ROCM_SDK_BASE_URL="${ROCM_SDK_RELEASE_BASE_URL}" + else + # Listing URL and tarball base share the same path; normalize trailing slash only. + ROCM_SDK_BASE_URL="${ROCM_SDK_RELEASE_URL%/}" + fi + fi + else + ROCM_SDK_BASE_URL="${ROCM_SDK_NIGHTLY_BASE_URL:-${ROCM_SDK_BASE_URL:-$_ROCM_NIGHTLY_BASE_DEFAULT}}" + fi + ROCM_SDK_INDEX_URL="${ROCM_SDK_NIGHTLY_INDEX_URL:-${ROCM_SDK_INDEX_URL:-$_ROCM_NIGHTLY_INDEX_DEFAULT}}" +fi + +# Post-build upload configuration +# Set UPLOAD_TARGET to enable automatic upload of built packages after the build. +# Packages are organized into: //// +# +# Supported formats: +# Local path: UPLOAD_TARGET="/mnt/shared/packages/" +# SCP: UPLOAD_TARGET="scp://user@buildserver:/opt/packages/" +# Rsync: UPLOAD_TARGET="rsync://user@buildserver:/opt/packages/" +# HTTP PUT: UPLOAD_TARGET="http://localhost:8080" +# (use with packages_server/ nginx setup for auto directory creation) +# +# Internal-only convenience: if UPLOAD_TARGET is unset, set RVS_AUTO_DETECT_LOCAL_UPLOAD=1 +# (e.g. in workflow env or shell) to probe http://localhost:8080/ once and use it when +# something responds. Default is off so external/CI builds never curl localhost; GitHub +# uploads use the workflow S3 steps, not Step 6 of this script. +# +# UPLOAD_REPO overrides the repo name in the upload path (auto-detected from git remote). +if [ -z "${UPLOAD_TARGET:-}" ] && [ -n "${RVS_AUTO_DETECT_LOCAL_UPLOAD:-}" ]; then + if curl -s -o /dev/null -w '' http://localhost:8080/ 2>/dev/null; then + UPLOAD_TARGET="http://localhost:8080" + fi +fi +UPLOAD_TARGET="${UPLOAD_TARGET:-}" +UPLOAD_REPO="${UPLOAD_REPO:-}" + +# Colors for output +RED='\033[0;31m' +GREEN='\033[0;32m' +YELLOW='\033[1;33m' +BLUE='\033[0;34m' +NC='\033[0m' # No Color + +print_info() { + echo -e "${BLUE}[INFO]${NC} $1" >&2 +} + +print_success() { + echo -e "${GREEN}[SUCCESS]${NC} $1" >&2 +} + +print_warning() { + echo -e "${YELLOW}[WARNING]${NC} $1" >&2 +} + +print_error() { + echo -e "${RED}[ERROR]${NC} $1" >&2 +} + +# Map common truthy/falsey strings to CMake ON/OFF. +normalize_on_off() { + case "$(echo "$1" | tr '[:upper:]' '[:lower:]')" in + 1|on|true|yes) echo ON ;; + *) echo OFF ;; + esac +} + +# Locate HIP device bitcode (amdgcn/bitcode) under a TheRock/ROCm SDK root. +# Probes known layouts in priority order — no find|head pipeline (fragile with set -e). +# Prints the path on stdout; returns 0 on success, 1 if not found. +resolve_hip_device_lib_path() { + local root="$1" + local cand resource_dir + + if [ -z "$root" ] || [ ! -d "$root" ]; then + return 1 + fi + + for cand in \ + "${root}/lib/llvm/amdgcn/bitcode" \ + "${root}/amdgcn/bitcode" \ + "${root}/lib/clang/amdgcn/bitcode" + do + if [ -d "$cand" ]; then + echo "$cand" + return 0 + fi + done + + # ROCm 7.2+ Clang resource-dir layout: lib/llvm/lib/clang//lib/amdgcn/bitcode + for cand in "${root}"/lib/llvm/lib/clang/*/lib/amdgcn/bitcode; do + if [ -d "$cand" ]; then + echo "$cand" + return 0 + fi + done + + # Derive from amdclang++ resource dir when present. + if [ -x "${root}/bin/amdclang++" ]; then + resource_dir="$("${root}/bin/amdclang++" -print-resource-dir 2>/dev/null)" || true + if [ -n "$resource_dir" ] && [ -d "${resource_dir}/lib/amdgcn/bitcode" ]; then + echo "${resource_dir}/lib/amdgcn/bitcode" + return 0 + fi + fi + + return 1 +} + +# Print diagnostics when resolve_hip_device_lib_path fails. +report_hip_device_lib_path_failure() { + local root="$1" + print_error "Could not find amdgcn/bitcode under ROCM_PATH=$root" + print_error "HIP device libraries are required for GPU code compilation" + print_error "Tried:" + print_error " ${root}/lib/llvm/amdgcn/bitcode" + print_error " ${root}/amdgcn/bitcode" + print_error " ${root}/lib/clang/amdgcn/bitcode" + print_error " ${root}/lib/llvm/lib/clang/*/lib/amdgcn/bitcode" + print_error " amdclang++ -print-resource-dir .../lib/amdgcn/bitcode" + if [ -d "$root" ]; then + print_error "Top-level contents of $root:" + ls -la "$root" 2>&1 | while read -r line; do print_error " $line"; done + if [ -d "${root}/lib/llvm" ]; then + print_error "Contents of ${root}/lib/llvm:" + ls -la "${root}/lib/llvm" 2>&1 | while read -r line; do print_error " $line"; done + fi + fi + print_error "If the SDK is missing amdgcn/bitcode, this may be a known TheRock symlink issue;" + print_error "see https://github.com/ROCm/TheRock/issues/74" +} + +# Join base URL and filename without duplicate slashes +join_base_and_file() { + local base="$1" + local path="$2" + base="${base%/}" + printf '%s/%s' "$base" "$path" +} + +# After ROCM_VERSION is known: pick tarball host directory from version shape (overrides channel defaults). +# Nightly: x.y.za Release: X.Y.Z (three-part semver only) +apply_sdk_tarball_base_for_version() { + local v="$1" + if echo "$v" | grep -qE '^[0-9]+\.[0-9]+\.[0-9]+a[0-9]+'; then + ROCM_SDK_BASE_URL="${ROCM_SDK_NIGHTLY_BASE_URL:-${ROCM_SDK_BASE_URL:-$_ROCM_NIGHTLY_BASE_DEFAULT}}" + print_info "Version matches ROCm nightly format (x.y.za…) → tarball base: $ROCM_SDK_BASE_URL" + elif echo "$v" | grep -qE '^[0-9]+\.[0-9]+\.[0-9]+$'; then + ROCM_SDK_BASE_URL="${ROCM_SDK_RELEASE_BASE_URL:-${ROCM_SDK_BASE_URL:-$_ROCM_RELEASE_BASE_DEFAULT}}" + print_info "Version matches ROCm release format (X.Y.Z) → tarball base: $ROCM_SDK_BASE_URL" + fi +} + +# Validate ROCM_VERSION after parsing an SDK listing (not -tests- or other variants). +# mode: nightly (x.y.za) or release (X.Y.Z) +validate_rocm_version_string() { + local v="$1" + local mode="$2" + case "$mode" in + nightly) + if echo "$v" | grep -qE '^[0-9]+\.[0-9]+\.[0-9]+a[0-9]+$'; then + return 0 + fi + ;; + release) + if echo "$v" | grep -qE '^[0-9]+\.[0-9]+\.[0-9]+$'; then + return 0 + fi + ;; + *) + print_error "validate_rocm_version_string: unknown mode $mode" + return 1 + ;; + esac + print_error "Invalid ROCm $mode version string from listing: $v" + return 1 +} + +# Parse SDK tarball versions from a listing on stdin (HTML/JSON). +# Per-match only — do not grep -v '-tests-' on the whole page (nightly index is one JSON line). +# Prints one version string per matching SDK tarball. +parse_sdk_tarball_versions_from_listing() { + local gpu_family="$1" + local mode="$2" + local pattern + + case "$mode" in + nightly) + # Digit after gpu_family- excludes ...-gfx110X-all-tests-... + pattern="therock-dist-linux-${gpu_family}-[0-9]+\.[0-9]+\.[0-9]+a[0-9]+" + ;; + release) + pattern="therock-dist-linux-${gpu_family}-[0-9]+\.[0-9]+\.[0-9]+" + ;; + *) + print_error "parse_sdk_tarball_versions_from_listing: unknown mode $mode" + return 1 + ;; + esac + + grep -oE "$pattern" | sed "s|^therock-dist-linux-${gpu_family}-||" +} + +# Collect SDK version strings from a listing file; python3 fallback when grep finds nothing. +collect_sdk_versions_from_listing_file() { + local listing_file="$1" + local gpu_family="$2" + local mode="$3" + local versions + + versions=$(parse_sdk_tarball_versions_from_listing "$gpu_family" "$mode" < "$listing_file") + + if [ -z "$versions" ] && [ -s "$listing_file" ] && command -v python3 >/dev/null 2>&1; then + versions=$(python3 - "$gpu_family" "$mode" "$listing_file" <<'PY' +import re +import sys + +gpu_family = sys.argv[1] +mode = sys.argv[2] +with open(sys.argv[3], encoding="utf-8", errors="replace") as f: + text = f.read() +prefix = "therock-dist-linux-" + re.escape(gpu_family) + "-" +if mode == "nightly": + pat = re.compile(prefix + r"([0-9]+\.[0-9]+\.[0-9]+a[0-9]+)\.tar\.gz") +else: + pat = re.compile(prefix + r"([0-9]+\.[0-9]+\.[0-9]+)\.tar\.gz") +seen = set() +for m in pat.finditer(text): + if "-tests-" in m.group(0): + continue + v = m.group(1) + if v not in seen: + seen.add(v) + print(v) +PY +) + if [ -n "$versions" ]; then + print_info "Parsed SDK versions via python3 fallback" + fi + fi + + printf '%s\n' "$versions" +} + +# Function to fetch latest ROCm version for specified GPU family +# Sets the global ROCM_VERSION variable directly +fetch_latest_rocm_version() { + local gpu_family="$1" + local listing_tmp latest_version listing_bytes + + print_info "Fetching latest ROCm version for $gpu_family..." + print_info "Index URL: $ROCM_SDK_INDEX_URL" + + listing_tmp="$(mktemp)" + if ! wget -q -O "$listing_tmp" "$ROCM_SDK_INDEX_URL"; then + print_error "wget failed for $ROCM_SDK_INDEX_URL" + rm -f "$listing_tmp" + return 1 + fi + if [ ! -s "$listing_tmp" ]; then + print_error "Empty listing from $ROCM_SDK_INDEX_URL" + rm -f "$listing_tmp" + return 1 + fi + + listing_bytes=$(wc -c < "$listing_tmp" | tr -d ' ') + latest_version=$(collect_sdk_versions_from_listing_file "$listing_tmp" "$gpu_family" nightly \ + | sort -V | tail -1) + rm -f "$listing_tmp" + + if [ -z "$latest_version" ]; then + print_error "Could not fetch latest ROCm nightly version for $gpu_family from $ROCM_SDK_INDEX_URL" + print_error "Listing was ${listing_bytes} bytes but no SDK tarballs matched therock-dist-linux-${gpu_family}-" + print_error "(-tests- tarballs are excluded; do not filter the listing with grep -v on whole lines)" + return 1 + fi + + if ! validate_rocm_version_string "$latest_version" nightly; then + return 1 + fi + + print_success "Found latest ROCm version: $latest_version" + ROCM_VERSION="$latest_version" + return 0 +} + +# Fetch latest ROCm *release* tarball version (semantic X.Y.Z only; excludes nightly builds like 7.11.0a20260121). +# Parses ROCM_SDK_RELEASE_URL (HTML listing / index). Sets ROCM_VERSION. +fetch_latest_rocm_release_version() { + local gpu_family="$1" + local list_url="${ROCM_SDK_RELEASE_URL:-$_ROCM_RELEASE_LIST_DEFAULT}" + local listing_tmp latest_version listing_bytes + + print_info "Fetching latest ROCm release version (X.Y.Z) for $gpu_family..." + print_info "Release listing URL: $list_url" + + listing_tmp="$(mktemp)" + if ! wget -q -O "$listing_tmp" "$list_url"; then + print_error "wget failed for $list_url" + rm -f "$listing_tmp" + return 1 + fi + if [ ! -s "$listing_tmp" ]; then + print_error "Empty listing from $list_url" + rm -f "$listing_tmp" + return 1 + fi + + listing_bytes=$(wc -c < "$listing_tmp" | tr -d ' ') + latest_version=$(collect_sdk_versions_from_listing_file "$listing_tmp" "$gpu_family" release \ + | sort -V | uniq | tail -1) + rm -f "$listing_tmp" + + if [ -z "$latest_version" ]; then + print_error "No therock-dist-linux-${gpu_family}-X.Y.Z.tar.gz found under release listing (expected forms like 7.11.0)" + print_error "Listing was ${listing_bytes} bytes but no SDK tarballs matched for $gpu_family" + return 1 + fi + + if ! validate_rocm_version_string "$latest_version" release; then + return 1 + fi + + print_success "Found latest ROCm release version: $latest_version" + ROCM_VERSION="$latest_version" + return 0 +} + +# ROCm SDK CMake configs (e.g. hipblaslt-config.cmake) call block(), which was added +# in CMake 3.25. Ubuntu 22.04 ships cmake 3.22 — too old. +# Try Kitware's APT repo first (clean apt integration); fall back to pip for restricted +# networks. Set RVS_SKIP_KITWARE_REPO=1 to skip the Kitware step (e.g. air-gapped runners). +ensure_recent_cmake_ubuntu() { + [ -f /etc/os-release ] || return 0 + # shellcheck source=/dev/null + . /etc/os-release + case "${ID:-}" in + ubuntu|debian) ;; + *) return 0 ;; + esac + command -v apt-get >/dev/null 2>&1 || return 0 + + local need_major=3 need_minor=25 + local cur_major=0 cur_minor=0 v="" + if command -v cmake >/dev/null 2>&1; then + v="$(cmake --version 2>/dev/null | head -1 | awk '{print $3}')" + cur_major="${v%%.*}" + local _rest="${v#*.}" + cur_minor="${_rest%%.*}" + cur_major="${cur_major:-0}" + cur_minor="${cur_minor:-0}" + if { [ "$cur_major" -gt "$need_major" ] 2>/dev/null; } \ + || { [ "$cur_major" -eq "$need_major" ] 2>/dev/null && [ "$cur_minor" -ge "$need_minor" ] 2>/dev/null; }; then + print_success "cmake $v is recent enough (>= ${need_major}.${need_minor})" + return 0 + fi + print_info "cmake $v is older than ${need_major}.${need_minor}; upgrading (ROCm hipblaslt-config uses block())" + else + print_info "cmake not installed; installing recent version (>= ${need_major}.${need_minor})" + fi + + # Try Kitware APT repo first + if [ -n "${VERSION_CODENAME:-}" ] && [ "${RVS_SKIP_KITWARE_REPO:-}" != "1" ]; then + print_info "Trying Kitware APT repo for cmake (codename: ${VERSION_CODENAME})..." + apt-get install -y --no-install-recommends ca-certificates gnupg wget >/dev/null 2>&1 || true + if wget -qO - https://apt.kitware.com/keys/kitware-archive-latest.asc 2>/dev/null \ + | gpg --dearmor -o /usr/share/keyrings/kitware-archive-keyring.gpg 2>/dev/null \ + && [ -s /usr/share/keyrings/kitware-archive-keyring.gpg ]; then + echo "deb [signed-by=/usr/share/keyrings/kitware-archive-keyring.gpg] https://apt.kitware.com/ubuntu/ ${VERSION_CODENAME} main" \ + > /etc/apt/sources.list.d/kitware.list + if apt-get update >/dev/null 2>&1 && apt-get install -y --no-install-recommends cmake; then + hash -r + local v_new + v_new="$(cmake --version 2>/dev/null | head -1 | awk '{print $3}')" + print_success "Installed cmake $v_new from Kitware APT repo" + return 0 + fi + print_warning "Kitware APT install failed; falling back to pip" + else + print_warning "Could not reach apt.kitware.com; falling back to pip (set RVS_SKIP_KITWARE_REPO=1 to silence this)" + fi + fi + + # pip fallback: works regardless of distro repo state, needs PyPI access. + if ! command -v pip3 >/dev/null 2>&1; then + apt-get install -y --no-install-recommends python3-pip >/dev/null 2>&1 || true + fi + if command -v pip3 >/dev/null 2>&1; then + if pip3 install --break-system-packages --upgrade cmake 2>/dev/null \ + || pip3 install --upgrade cmake; then + hash -r + local v_new + v_new="$(cmake --version 2>/dev/null | head -1 | awk '{print $3}')" + print_success "Installed cmake $v_new via pip" + return 0 + fi + fi + + print_error "Could not install cmake >= ${need_major}.${need_minor}; ROCm hipblaslt configure will fail." + print_error "Either expose apt.kitware.com / PyPI to this runner, or pre-install cmake>=3.25 in the image." + return 1 +} + +# Function to check and install dependencies +check_and_install_dependencies() { + print_info "Checking for required build dependencies..." + + # Detect OS + if [ -f /etc/os-release ]; then + . /etc/os-release + OS=$ID + else + print_error "Cannot detect OS. Please install dependencies manually." + exit 1 + fi + + # Check for command-line tools + MISSING_TOOLS=() + command -v cmake >/dev/null 2>&1 || MISSING_TOOLS+=("cmake") + command -v make >/dev/null 2>&1 || MISSING_TOOLS+=("make") + command -v gcc >/dev/null 2>&1 || MISSING_TOOLS+=("gcc") + command -v g++ >/dev/null 2>&1 || MISSING_TOOLS+=("g++") + command -v git >/dev/null 2>&1 || MISSING_TOOLS+=("git") + command -v wget >/dev/null 2>&1 || MISSING_TOOLS+=("wget") + command -v tar >/dev/null 2>&1 || MISSING_TOOLS+=("tar") + command -v doxygen >/dev/null 2>&1 || MISSING_TOOLS+=("doxygen") + command -v python3 >/dev/null 2>&1 || MISSING_TOOLS+=("python3") + command -v patchelf >/dev/null 2>&1 || MISSING_TOOLS+=("patchelf") + + # Check for library dependencies (platform-specific) + MISSING_LIBS=() + if [[ "$OS" =~ ^(ubuntu|debian)$ ]]; then + # Check for Ubuntu/Debian library headers + [ -f /usr/include/pci/pci.h ] || MISSING_LIBS+=("libpci-dev") + [ -f /usr/include/yaml-cpp/yaml.h ] || MISSING_LIBS+=("libyaml-cpp-dev") + [ -f /usr/include/numa.h ] || MISSING_LIBS+=("libnuma-dev") + command -v rpmbuild >/dev/null 2>&1 || MISSING_LIBS+=("rpm") + command -v unzip >/dev/null 2>&1 || MISSING_LIBS+=("unzip") + elif [[ "$OS" =~ ^(centos|rhel|rocky|almalinux|amzn)$ ]]; then + # Check for CentOS/RHEL/Rocky/AlmaLinux library headers + [ -f /usr/include/pci/pci.h ] || MISSING_LIBS+=("pciutils-devel") + [ -f /usr/include/yaml-cpp/yaml.h ] || MISSING_LIBS+=("yaml-cpp-devel") + [ -f /usr/include/numa.h ] || MISSING_LIBS+=("numactl-devel") + command -v rpmbuild >/dev/null 2>&1 || MISSING_LIBS+=("rpm-build") + elif [[ "$OS" =~ ^(sles|opensuse-leap|opensuse-tumbleweed)$ ]]; then + # Check for SUSE library headers (libnuma-devel on SLES 15/16, not numactl-devel) + [ -f /usr/include/pci/pci.h ] || MISSING_LIBS+=("pciutils-devel") + [ -f /usr/include/yaml-cpp/yaml.h ] || MISSING_LIBS+=("yaml-cpp-devel") + [ -f /usr/include/numa.h ] || MISSING_LIBS+=("libnuma-devel") + command -v rpmbuild >/dev/null 2>&1 || MISSING_LIBS+=("rpm-build") + fi + + # Combine missing tools and libraries + MISSING_DEPS=() + [ ${#MISSING_TOOLS[@]} -ne 0 ] && MISSING_DEPS+=("${MISSING_TOOLS[@]}") + [ ${#MISSING_LIBS[@]} -ne 0 ] && MISSING_DEPS+=("${MISSING_LIBS[@]}") + + if [ ${#MISSING_DEPS[@]} -ne 0 ]; then + print_warning "Missing dependencies: ${MISSING_DEPS[*]}" + echo "" + + if [[ "$OS" =~ ^(ubuntu|debian)$ ]]; then + print_info "Installing dependencies for Ubuntu/Debian..." + apt-get update + apt-get install -y \ + build-essential \ + cmake \ + git \ + wget \ + libpci3 \ + libpci-dev \ + doxygen \ + unzip \ + libyaml-cpp-dev \ + rpm \ + python3 \ + libnuma-dev \ + patchelf + elif [[ "$OS" =~ ^(centos|rhel|rocky|almalinux|amzn)$ ]]; then + print_info "Installing dependencies for CentOS/RHEL/Rocky/AlmaLinux..." + + # Check if running in manylinux container (has /opt/python) + if [ -d /opt/python ]; then + print_info "Detected manylinux environment - some tools may be pre-installed" + fi + + # Enable PowerTools/CRB repository (contains doxygen and yaml-cpp) + print_info "Enabling PowerTools/CRB repository..." + if command -v dnf >/dev/null 2>&1; then + # AlmaLinux/Rocky Linux 8+ uses dnf + dnf install -y dnf-plugins-core 2>/dev/null || true + dnf config-manager --set-enabled powertools 2>/dev/null || \ + dnf config-manager --set-enabled crb 2>/dev/null || \ + dnf config-manager --set-enabled devel 2>/dev/null || \ + print_warning "Could not enable PowerTools/CRB/devel repo (may already be enabled)" + else + # CentOS 8 uses yum + yum install -y yum-utils 2>/dev/null || true + yum-config-manager --enable powertools 2>/dev/null || \ + yum-config-manager --enable PowerTools 2>/dev/null || \ + yum-config-manager --enable devel 2>/dev/null || \ + print_warning "Could not enable PowerTools/devel repo (may already be enabled)" + fi + + # Install EPEL repository (may already be present in manylinux) + print_info "Installing EPEL repository..." + yum install -y epel-release 2>/dev/null || print_warning "EPEL may already be installed" + + print_info "Installing build dependencies..." + # Note: manylinux typically has gcc, g++, make pre-installed + yum install -y \ + gcc \ + gcc-c++ \ + make \ + git \ + wget \ + tar \ + pciutils-devel \ + doxygen \ + rpm-build \ + python3 \ + numactl-devel \ + patchelf \ + || print_warning "Some packages may already be installed" + + # Install a gcc-toolset with C++20 support (requires GCC >= 11) + # Try highest available first for best C++20/C++23 support, fall back to minimum viable + # RHEL/AlmaLinux/Rocky 8+ use gcc-toolset-{ver}, CentOS 7 uses devtoolset-{ver} + if [[ "$OS" =~ ^(centos|rhel|almalinux|rocky)$ ]]; then + GCC_TOOLSET_INSTALLED="" + for ver in 14 13 12 11; do + if yum list available gcc-toolset-${ver} &>/dev/null; then + print_info "Found gcc-toolset-${ver} - installing for C++20 support..." + if yum install -y gcc-toolset-${ver}; then + GCC_TOOLSET_INSTALLED="gcc-toolset-${ver}" + print_success "Installed gcc-toolset-${ver} (C++20 supported)" + break + fi + fi + done + # Fallback to devtoolset (CentOS 7) - only devtoolset-11 has + if [ -z "$GCC_TOOLSET_INSTALLED" ]; then + if yum list available devtoolset-11 &>/dev/null; then + print_info "Found devtoolset-11 - installing for C++20 support (CentOS 7)..." + if yum install -y devtoolset-11; then + GCC_TOOLSET_INSTALLED="devtoolset-11" + print_success "Installed devtoolset-11 (C++20 supported)" + fi + fi + fi + if [ -z "$GCC_TOOLSET_INSTALLED" ]; then + print_warning "No gcc-toolset (11-14) or devtoolset-11 available - C++20 headers like may be missing" + fi + fi + + # Install cmake and yaml-cpp separately as they may need special handling + print_info "Installing cmake..." + yum install -y cmake3 || yum install -y cmake || print_warning "cmake installation may have failed" + + print_info "Installing yaml-cpp..." + yum install -y yaml-cpp-devel yaml-cpp-static 2>/dev/null || \ + print_warning "yaml-cpp may not be available - will try to continue" + elif [[ "$OS" =~ ^(sles|opensuse-leap|opensuse-tumbleweed)$ ]]; then + print_info "Installing dependencies for SUSE/SLES..." + zypper --non-interactive refresh || print_warning "zypper refresh reported errors; continuing" + zypper --non-interactive install -y \ + gcc \ + gcc-c++ \ + make \ + git \ + wget \ + tar \ + cmake \ + doxygen \ + python3 \ + pciutils-devel \ + libpci3 \ + yaml-cpp-devel \ + libnuma-devel \ + rpm \ + rpm-build \ + patchelf \ + || print_warning "Some packages may already be installed" + else + print_error "Unsupported OS: $OS" + echo "" + echo "Please install the following dependencies manually:" + echo "" + echo "Build Tools:" + echo " - gcc, g++, make" + echo " - cmake" + echo " - git" + echo " - wget" + echo " - tar" + echo " - doxygen" + echo " - python3" + echo "" + echo "Development Libraries:" + echo " - libpci-dev (or pciutils-devel)" + echo " - libyaml-cpp-dev (or yaml-cpp-devel)" + echo " - libnuma-dev (Ubuntu), numactl-devel (RHEL/CentOS), or libnuma-devel (SLES)" + echo " - patchelf (CPack RUNPATH normalization for DEB/RPM/TGZ)" + echo " - rpm-build tools" + exit 1 + fi + + print_success "Dependencies installed successfully" + echo "" + + # Verify installation by re-checking + print_info "Verifying installation..." + VERIFY_FAILED=() + for tool in "${MISSING_TOOLS[@]}"; do + if ! command -v "$tool" >/dev/null 2>&1; then + VERIFY_FAILED+=("$tool") + fi + done + + if [ ${#VERIFY_FAILED[@]} -ne 0 ]; then + print_error "Failed to install: ${VERIFY_FAILED[*]}" + exit 1 + fi + print_success "All dependencies verified" + else + print_success "All required dependencies found" + fi + echo "" +} + +# Check and install dependencies (installs wget, etc. - required before fetch_latest_rocm_version) +check_and_install_dependencies +ensure_recent_cmake_ubuntu + +# Determine ROCm version: explicit env > channel-specific listing +if [ -n "$ROCM_VERSION" ]; then + print_info "Using specified ROCm version: $ROCM_VERSION" +elif [ "$ROCM_SDK_CHANNEL" = "nightly" ]; then + print_info "No ROCm version specified; fetching latest from ROCm nightly index (channel=nightly)..." + if [ -z "$ROCM_SDK_INDEX_URL" ]; then + print_error "ROCM_SDK_CHANNEL=nightly requires ROCM_SDK_INDEX_URL" + exit 1 + fi + fetch_latest_rocm_version "$GPU_FAMILY" + if [ -z "$ROCM_VERSION" ]; then + print_error "Failed to determine ROCm nightly version" + exit 1 + fi +elif [ "$ROCM_SDK_CHANNEL" = "release" ]; then + print_info "No ROCm version specified; fetching latest release X.Y.Z from ROCM_SDK_RELEASE_URL..." + if [ -z "$ROCM_SDK_RELEASE_URL" ]; then + print_error "ROCM_SDK_CHANNEL=release requires ROCM_SDK_RELEASE_URL when ROCM_VERSION is unset" + exit 1 + fi + fetch_latest_rocm_release_version "$GPU_FAMILY" + if [ -z "$ROCM_VERSION" ]; then + print_error "Failed to determine ROCm release version" + exit 1 + fi +elif [ -n "$ROCM_SDK_RELEASE_URL" ]; then + print_info "No ROCm version specified; resolving latest release from ROCM_SDK_RELEASE_URL (channel=auto)..." + fetch_latest_rocm_release_version "$GPU_FAMILY" + if [ -z "$ROCM_VERSION" ]; then + print_error "Failed to determine ROCm release version" + exit 1 + fi +elif [ -n "$ROCM_SDK_INDEX_URL" ]; then + print_info "No ROCm version specified, fetching latest from nightly index (channel=auto)..." + fetch_latest_rocm_version "$GPU_FAMILY" + + if [ -z "$ROCM_VERSION" ]; then + print_error "Failed to determine ROCm version" + exit 1 + fi +else + print_error "ROCM_VERSION is required" + print_info "Set ROCM_VERSION, or configure ROCM_SDK_RELEASE_URL (release X.Y.Z tarballs) or ROCM_SDK_INDEX_URL (nightly listing)." + print_info "Example:" + echo "" + echo " export ROCM_VERSION=\"7.11.0\"" + echo " ./build_packages_local.sh" + echo "" + exit 1 +fi + +apply_sdk_tarball_base_for_version "$ROCM_VERSION" + +# Publish the resolved version to GITHUB_ENV so downstream jobs (e.g. +# publish-unsigned-latest) can read the actual nightly version rather than +# the static vars.ROCM_VERSION repository variable. +if [ -n "${GITHUB_ENV:-}" ]; then + echo "ROCM_VERSION=${ROCM_VERSION}" >> "$GITHUB_ENV" +fi + +BUILD_TRANSFERBENCH_CLI="$(normalize_on_off "$BUILD_TRANSFERBENCH_CLI")" + +# Print configuration +echo "================================================================================" +echo " ROCm Validation Suite - Local Package Build Script" +echo "================================================================================" +print_info "ROCm Version: $ROCM_VERSION" +print_info "ROCm SDK channel: ${ROCM_SDK_CHANNEL:-auto}" +print_info "GPU Family: $GPU_FAMILY" +print_info "Build Type: $BUILD_TYPE" +print_info "Build TransferBench CLI: $BUILD_TRANSFERBENCH_CLI" +if [ "$BUILD_TRANSFERBENCH_CLI" = "ON" ]; then + print_info "TransferBench GPU targets: $TRANSFERBENCH_GPU_TARGETS" +fi +print_info "Build Directory: $BUILD_DIR" +print_info "ROCm Install: $ROCM_INSTALL_DIR" +print_info "SDK Source: $ROCM_SDK_BASE_URL" +if [ -n "$UPLOAD_TARGET" ]; then + print_info "Upload Target: $UPLOAD_TARGET" +fi +print_info "RVS Version: Will be read from CMakeLists.txt by CMake/CPack" +echo "================================================================================" +echo "" + +# Step 1: Download ROCm SDK +print_info "Step 1: Downloading ROCm SDK tarball..." +TARBALL_URL=$(join_base_and_file "$ROCM_SDK_BASE_URL" "therock-dist-linux-${GPU_FAMILY}-${ROCM_VERSION}.tar.gz") +TARBALL_FILE="$ROCM_INSTALL_DIR/rocm-sdk.tar.gz" + +mkdir -p "$ROCM_INSTALL_DIR" + +if [ -f "$ROCM_INSTALL_DIR/install/bin/hipconfig" ]; then + print_warning "ROCm SDK already exists at $ROCM_INSTALL_DIR/install" + if [ -t 0 ]; then + read -p "Do you want to re-download? (y/N): " -n 1 -r + echo + if [[ $REPLY =~ ^[Yy]$ ]]; then + rm -rf "$ROCM_INSTALL_DIR/install" + else + print_info "Using existing ROCm SDK" + export ROCM_PATH="$ROCM_INSTALL_DIR/install" + print_success "ROCm SDK path set to: $ROCM_PATH" + echo "" + goto_step2=true + fi + else + print_info "Non-interactive mode: reusing existing ROCm SDK" + export ROCM_PATH="$ROCM_INSTALL_DIR/install" + print_success "ROCm SDK path set to: $ROCM_PATH" + echo "" + goto_step2=true + fi +fi + +if [ -z "$goto_step2" ]; then + print_info "Downloading from: $TARBALL_URL" + if wget --spider "$TARBALL_URL" 2>/dev/null; then + # Avoid flooding CI logs with per-chunk progress on multi-GB SDK tarballs. + if [ -t 1 ]; then + wget --show-progress -O "$TARBALL_FILE" "$TARBALL_URL" + else + wget -q -O "$TARBALL_FILE" "$TARBALL_URL" + fi + if [ -f "$TARBALL_FILE" ]; then + print_success "Download complete ($(du -h "$TARBALL_FILE" | awk '{print $1}'))" + else + print_success "Download complete" + fi + else + print_error "Failed to download ROCm SDK tarball" + print_error "URL: $TARBALL_URL" + print_info "Please check if the ROCm version and GPU family are correct" + exit 1 + fi + + # Extract tarball + print_info "Extracting ROCm SDK..." + mkdir -p "$ROCM_INSTALL_DIR/install" + tar -xzf "$TARBALL_FILE" -C "$ROCM_INSTALL_DIR/install" --strip-components=1 + print_success "Extraction complete" + + export ROCM_PATH="$ROCM_INSTALL_DIR/install" + print_success "ROCm SDK installed to: $ROCM_PATH" +fi +echo "" + +# Step 2: Setup environment +print_info "Step 2: Setting up ROCm environment..." + +# Find HIP device library path (amdgcn/bitcode) first +print_info "Locating HIP device libraries (amdgcn/bitcode)..." +HIP_DEVICE_LIB_PATH="" +if ! HIP_DEVICE_LIB_PATH="$(resolve_hip_device_lib_path "$ROCM_PATH")"; then + report_hip_device_lib_path_failure "$ROCM_PATH" + exit 1 +fi +print_success "HIP_DEVICE_LIB_PATH=$HIP_DEVICE_LIB_PATH" + +# Export all environment variables +export PATH="$ROCM_PATH/bin:$PATH" +export LD_LIBRARY_PATH="$ROCM_PATH/lib:$LD_LIBRARY_PATH" +export CMAKE_PREFIX_PATH="$ROCM_PATH:$CMAKE_PREFIX_PATH" +export HIP_DEVICE_LIB_PATH="$HIP_DEVICE_LIB_PATH" + +# Extract major.minor version from ROCM_VERSION for ROCM_LIBPATCH_VERSION +# Convert to xxyy format with zero padding +# Example: "7.11.0a20260121" -> "0711", "8.0.0" -> "0800", "10.2.0" -> "1002" +ROCM_VERSION_MAJOR_MINOR=$(echo "$ROCM_VERSION" | sed -nE 's/^([0-9]+\.[0-9]+).*/\1/p') +if [ -z "$ROCM_VERSION_MAJOR_MINOR" ]; then + print_error "Could not extract major.minor version from ROCM_VERSION: $ROCM_VERSION" + exit 1 +fi + +# Split into major and minor, zero-pad to 2 digits each +ROCM_MAJOR=$(echo "$ROCM_VERSION_MAJOR_MINOR" | cut -d'.' -f1) +ROCM_MINOR=$(echo "$ROCM_VERSION_MAJOR_MINOR" | cut -d'.' -f2) +ROCM_LIBPATCH_VERSION=$(printf "%02d%02d" "$ROCM_MAJOR" "$ROCM_MINOR") + +export ROCM_LIBPATCH_VERSION +export ROCM_MAJOR +print_success "Set ROCM_LIBPATCH_VERSION=$ROCM_LIBPATCH_VERSION, ROCM_MAJOR=$ROCM_MAJOR (from $ROCM_VERSION_MAJOR_MINOR)" + +# Git 2.35+ refuses commands in container workspaces owned by a different UID than the +# checkout (Actions mounts ${{ github.workspace }} into the container as root). Mark the +# workspace as safe so subsequent git rev-parse calls below don't return "0000000". +if [ -n "${GITHUB_WORKSPACE:-}" ] && command -v git >/dev/null 2>&1; then + git config --global --add safe.directory "$GITHUB_WORKSPACE" 2>/dev/null || true +fi + +# Set CPACK package release based on event/branch type +# - Base (schedule, push, workflow_dispatch, local): r.yyyymmdd +# (ROCM_LIBPATCH_VERSION is major+minor as xxyy, e.g. 0711 for ROCm 7.11) +# - pull_request: same base + .branch.commit (source branch from GITHUB_HEAD_REF) +# - Release branches (starting with "rel"): GITHUB run number +# Prefer GITHUB_REF_NAME / GITHUB_HEAD_REF / GITHUB_SHA in Actions; fall back to git locally. +RELEASE_DATE="$(date +%Y%m%d)" +BASE_PACKAGE_RELEASE="r${ROCM_LIBPATCH_VERSION}.${RELEASE_DATE}" + +if [ "$GITHUB_EVENT_NAME" = "pull_request" ] && [ -n "${GITHUB_HEAD_REF:-}" ]; then + GIT_BRANCH="${GITHUB_HEAD_REF}" +else + GIT_BRANCH="${RVS_BUILD_REF_NAME:-${GITHUB_REF_NAME:-$(git rev-parse --abbrev-ref HEAD 2>/dev/null || echo "unknown")}}" +fi +if [ -n "$GITHUB_SHA" ]; then + GIT_COMMIT_SHORT="${GITHUB_SHA:0:7}" +else + GIT_COMMIT_SHORT=$(git rev-parse --short HEAD 2>/dev/null || echo "0000000") +fi +SANITIZED_BRANCH=$(echo "$GIT_BRANCH" | sed 's/[^A-Za-z0-9.+~]/./g') + +if [ "${GITHUB_EVENT_NAME}" != "pull_request" ] && [[ "$GIT_BRANCH" =~ ^rel ]]; then + GITHUB_RUN_NUMBER="${GITHUB_RUN_NUMBER:-1}" + PACKAGE_RELEASE="$GITHUB_RUN_NUMBER" + print_success "Set CPACK package release: $PACKAGE_RELEASE (release branch: $GIT_BRANCH, run: $GITHUB_RUN_NUMBER)" +elif [ "$GITHUB_EVENT_NAME" = "pull_request" ]; then + PACKAGE_RELEASE="${BASE_PACKAGE_RELEASE}.${SANITIZED_BRANCH}.${GIT_COMMIT_SHORT}" + print_success "Set CPACK package release: $PACKAGE_RELEASE (PR: $BASE_PACKAGE_RELEASE + $GIT_BRANCH + $GIT_COMMIT_SHORT)" +else + PACKAGE_RELEASE="$BASE_PACKAGE_RELEASE" + print_success "Set CPACK package release: $PACKAGE_RELEASE (r.date: ${BASE_PACKAGE_RELEASE}, event: ${GITHUB_EVENT_NAME:-local})" +fi + +export CPACK_DEBIAN_PACKAGE_RELEASE="$PACKAGE_RELEASE" +export CPACK_RPM_PACKAGE_RELEASE="$PACKAGE_RELEASE" + +if [[ "$OS" =~ ^(centos|rhel|almalinux|rocky)$ ]]; then + # Enable the installed gcc-toolset/devtoolset so hipcc (clang) finds C++20 headers like + # Discover the toolset path dynamically via rpm rather than hardcoding /opt/rh/ + + GCC_TOOLSET_ENABLED="" + for pkg in gcc-toolset-14 gcc-toolset-13 gcc-toolset-12 gcc-toolset-11 devtoolset-11; do + ENABLE_SCRIPT=$(rpm -ql ${pkg}-runtime 2>/dev/null | grep '/enable$' | head -1) + if [ -n "$ENABLE_SCRIPT" ] && [ -f "$ENABLE_SCRIPT" ]; then + TOOLSET_ROOT=$(dirname "$ENABLE_SCRIPT") + source "$ENABLE_SCRIPT" + export GCC_TOOLCHAIN="${TOOLSET_ROOT}/root/usr" + GCC_TOOLSET_ENABLED="$pkg" + print_success "Enabled ${pkg} from ${TOOLSET_ROOT} (GCC $(gcc -dumpversion))" + break + fi + done + if [ -z "$GCC_TOOLSET_ENABLED" ]; then + print_warning "No gcc-toolset (11-14) or devtoolset-11 found - C++20 headers may be missing" + fi + + if [ -x "$ROCM_PATH/bin/hipcc" ]; then + export CMAKE_CXX_COMPILER="$ROCM_PATH/bin/hipcc" + print_success "Set CMAKE_CXX_COMPILER to hipcc (CentOS/RHEL/AlmaLinux/Rocky)" + else + print_warning "hipcc not found at $ROCM_PATH/bin/hipcc - using system default compiler" + fi + + # Use cmake3 for RHEL-based distros + export CMAKE_COMMAND="cmake3" + print_info "Using cmake3 (CentOS/RHEL/AlmaLinux/Rocky)" +else + # Ubuntu/Debian: modules (pebb/pbqt) compile with hipcc, which is Clang-based and does not + # bundle libstdc++ C++20 headers such as . Pass the system GCC tree to hipcc via + # CMAKE_CXX_FLAGS --gcc-toolchain (same mechanism as GCC_TOOLCHAIN on RHEL). + export CMAKE_COMMAND="cmake" + print_info "Using cmake (Ubuntu/Debian/other)" + + if [[ "$OS" =~ ^(ubuntu|debian)$ ]]; then + if [ -x "$ROCM_PATH/bin/hipcc" ]; then + export CMAKE_CXX_COMPILER="$ROCM_PATH/bin/hipcc" + print_success "Set CMAKE_CXX_COMPILER to hipcc (Ubuntu/Debian)" + else + print_warning "hipcc not found at $ROCM_PATH/bin/hipcc - using CMake default C++ compiler" + fi + if compgen -G "/usr/include/c++/*/barrier" >/dev/null 2>&1; then + export GCC_TOOLCHAIN="/usr" + print_success "Set GCC_TOOLCHAIN=/usr so hipcc finds system libstdc++ (C++20 )" + else + print_warning "No /usr/include/c++/.../barrier found; install g++ 11+ (e.g. apt install g++-11 build-essential)" + fi + fi +fi + +print_info "Environment variables set:" +echo " ROCM_PATH=$ROCM_PATH" +echo " PATH includes: $ROCM_PATH/bin" +echo " LD_LIBRARY_PATH includes: $ROCM_PATH/lib" +echo " HIP_DEVICE_LIB_PATH=$HIP_DEVICE_LIB_PATH" +echo " ROCM_LIBPATCH_VERSION=$ROCM_LIBPATCH_VERSION" +if [ -n "$CPACK_DEBIAN_PACKAGE_RELEASE" ]; then + echo " CPACK_DEBIAN_PACKAGE_RELEASE=$CPACK_DEBIAN_PACKAGE_RELEASE" +fi +if [ -n "$CPACK_RPM_PACKAGE_RELEASE" ]; then + echo " CPACK_RPM_PACKAGE_RELEASE=$CPACK_RPM_PACKAGE_RELEASE" +fi +if [ -n "$CMAKE_CXX_COMPILER" ]; then + echo " CMAKE_CXX_COMPILER=$CMAKE_CXX_COMPILER" +fi +if [ -n "$GCC_TOOLCHAIN" ]; then + echo " GCC_TOOLCHAIN=$GCC_TOOLCHAIN" +fi +echo " CMAKE_COMMAND=$CMAKE_COMMAND" + +# Verify ROCm installation +print_info "Verifying ROCm installation..." +if [ -x "$ROCM_PATH/bin/hipconfig" ]; then + print_success "hipconfig found" +else + print_warning "hipconfig not found or not executable" +fi + +if [ -d "$ROCM_PATH/lib" ]; then + LIB_COUNT=$(ls -1 "$ROCM_PATH/lib"/*.so 2>/dev/null | wc -l) + print_success "Found $LIB_COUNT shared libraries in $ROCM_PATH/lib" +else + print_error "ROCm lib directory not found" + exit 1 +fi +echo "" + +# Step 3: Configure CMake +print_info "Step 3: Configuring CMake with relocatable paths..." +if [ -d "$BUILD_DIR" ]; then + print_warning "Build directory exists. Cleaning..." + rm -rf "$BUILD_DIR" +fi + +# Build cmake command with optional CXX compiler. +# CMAKE_INSTALL_RPATH / CMAKE_SKIP_RPATH / CMAKE_INSTALL_RPATH_USE_LINK_PATH live in CMakeLists.txt. +CMAKE_ARGS=( + -B "$BUILD_DIR" + -DCMAKE_BUILD_TYPE="$BUILD_TYPE" + -DROCM_PATH="$ROCM_PATH" + -DHIP_PLATFORM=amd + -DROCM_MAJOR_VERSION="$ROCM_MAJOR" + -DCPACK_PACKAGE_NAME="amdrocm${ROCM_MAJOR}-rvs" + -DCMAKE_INSTALL_PREFIX="/opt/rocm/extras-${ROCM_MAJOR}" + -DCPACK_PACKAGING_INSTALL_PREFIX="/opt/rocm/extras-${ROCM_MAJOR}" + -DCMAKE_VERBOSE_MAKEFILE=1 + -DFETCH_ROCMPATH_FROM_ROCMCORE=ON + -DBUILD_TRANSFERBENCH_CLI="$BUILD_TRANSFERBENCH_CLI" +) + +if [ "$BUILD_TRANSFERBENCH_CLI" = "ON" ]; then + CMAKE_ARGS+=(-DTRANSFERBENCH_GPU_TARGETS="$TRANSFERBENCH_GPU_TARGETS") +fi + +# Add CXX compiler if set +if [ -n "$CMAKE_CXX_COMPILER" ]; then + CMAKE_ARGS+=(-DCMAKE_CXX_COMPILER="$CMAKE_CXX_COMPILER") +fi + +# Point hipcc at the GCC toolchain for C++20 standard library headers +if [ -n "$GCC_TOOLCHAIN" ]; then + CMAKE_ARGS+=(-DCMAKE_CXX_FLAGS="--gcc-toolchain=$GCC_TOOLCHAIN") + print_success "Set --gcc-toolchain=$GCC_TOOLCHAIN for hipcc" +fi + +$CMAKE_COMMAND "${CMAKE_ARGS[@]}" + +if [ $? -eq 0 ]; then + print_success "CMake configuration successful" +else + print_error "CMake configuration failed" + exit 1 +fi +echo "" + +# Step 4: Build RVS +print_info "Step 4: Building RVS..." +NPROC=$(nproc 2>/dev/null || sysctl -n hw.ncpu 2>/dev/null || echo 4) +print_info "Using $NPROC parallel jobs" + +make -C "$BUILD_DIR" -j"$NPROC" + +if [ $? -eq 0 ]; then + print_success "Build successful" +else + print_error "Build failed" + exit 1 +fi +echo "" + +# Step 5: Create packages +print_info "Step 5: Creating packages..." + +# Create DEB package (Ubuntu/Debian) +if command -v dpkg >/dev/null 2>&1; then + print_info "Creating DEB package..." + cd "$BUILD_DIR" + cpack -G DEB --verbose + + if [ $? -eq 0 ]; then + DEB_FILE=$(ls amdrocm*-rvs*.deb 2>/dev/null | head -1) + if [ -n "$DEB_FILE" ]; then + print_success "Created DEB package: $DEB_FILE" + DEB_SIZE=$(du -h "$DEB_FILE" | cut -f1) + print_info "Package size: $DEB_SIZE" + + # Verify DEB package + print_info "Verifying DEB package..." + dpkg-deb -I "$DEB_FILE" | head -20 + fi + else + print_warning "DEB package creation failed (might be expected on non-Debian systems)" + fi + cd - > /dev/null +fi + +# Create RPM package (CentOS/RHEL/SLES) +if command -v rpm >/dev/null 2>&1; then + print_info "Creating RPM package..." + cd "$BUILD_DIR" + cpack -G RPM --verbose + + + if [ $? -eq 0 ]; then + RPM_FILE=$(ls amdrocm*-rvs*.rpm 2>/dev/null | head -1) + if [ -n "$RPM_FILE" ]; then + print_success "Created RPM package: $RPM_FILE" + RPM_SIZE=$(du -h "$RPM_FILE" | cut -f1) + print_info "Package size: $RPM_SIZE" + + # Verify RPM package + print_info "Verifying RPM package..." + rpm -qip "$RPM_FILE" | head -20 + fi + else + print_warning "RPM package creation failed (might be expected on non-RPM systems)" + fi + cd - > /dev/null +fi + +# Create TGZ package +# +# TGZ uses a different layout than DEB/RPM: CPACK_PACKAGING_INSTALL_PREFIX is +# emptied so files land at the tarball root (bin/, lib/, share/) rather than +# under /opt/rocm/extras-/. RVS bakes CPACK_PACKAGING_INSTALL_PREFIX into +# each install() DESTINATION at configure time, so we must RE-configure (not +# just re-cpack) for the destinations to flip. +# +# Two extra steps are needed for a clean tarball: +# - Re-build after the re-configure so ExternalProject sub-builds (the bundled +# TransferBench CLI) are rebuilt against the new state; otherwise cpack would +# copy from a stale sub-build tree and silently miss files. +# - Prune empty directories from the tarball afterwards. With +# CPACK_SET_DESTDIR=ON, CPack materialises CMAKE_INSTALL_PREFIX +# (/opt/rocm/extras-) under DESTDIR even though every real install() +# destination keys off the emptied CPACK_PACKAGING_INSTALL_PREFIX, leaving an +# empty opt/rocm/extras-/ tree. We must NOT fix this by setting +# CMAKE_INSTALL_PREFIX=/ : GNUInstallDirs' FHS special case then rewrites +# bin->usr/bin, lib->usr/lib, ... (and -DCMAKE_INSTALL_BINDIR=bin does not +# override it), nesting the whole package under usr/. Instead keep the root +# layout and strip empty dirs from the tarball after cpack. RVS never ships +# intentionally-empty directories. +print_info "Creating TGZ package..." +CMAKE_ARGS+=(-DCPACK_SET_DESTDIR=ON -DCPACK_MONOLITHIC_INSTALL=ON "-DCPACK_PACKAGING_INSTALL_PREFIX:PATH=" -DCPACK_INCLUDE_TOPLEVEL_DIRECTORY=OFF) +$CMAKE_COMMAND "${CMAKE_ARGS[@]}" +$CMAKE_COMMAND --build "$BUILD_DIR" -- -j"$(nproc 2>/dev/null || echo 4)" || print_warning "Re-build before TGZ packaging reported errors; proceeding with cpack" +cd "$BUILD_DIR" +cpack -G TGZ --verbose +cpack_tgz_status=$? + +# Strip empty directories left in the tarball by CPack DESTDIR staging +# (notably opt/rocm/extras-/). See the comment above the TGZ configure. +if [ $cpack_tgz_status -eq 0 ]; then + for _tgz in amdrocm*-rvs*.tar.gz; do + [ -e "$_tgz" ] || continue + _staged=$(mktemp -d) + if tar xzf "$_tgz" -C "$_staged"; then + find "$_staged" -depth -mindepth 1 -type d -empty -delete + ( cd "$_staged" && tar czf "$OLDPWD/$_tgz" -- * ) + print_info "Pruned empty staging directories from $_tgz" + fi + rm -rf "$_staged" + done +fi + +if [ $cpack_tgz_status -eq 0 ]; then + TGZ_FILE=$(ls amdrocm*-rvs*.tar.gz 2>/dev/null | head -1) + if [ -n "$TGZ_FILE" ]; then + print_success "Created TGZ package: $TGZ_FILE" + TGZ_SIZE=$(du -h "$TGZ_FILE" | cut -f1) + print_info "Package size: $TGZ_SIZE" + fi +else + print_error "TGZ package creation failed" +fi + +cd - > /dev/null +echo "" + +# Step 6: Upload packages (optional) +if [ -n "$UPLOAD_TARGET" ]; then + print_info "Step 6: Uploading packages to $UPLOAD_TARGET..." + + # Build organized upload subpath: // + if [ -z "$UPLOAD_REPO" ]; then + UPLOAD_REPO="${GITHUB_REPOSITORY##*/}" + [ -z "$UPLOAD_REPO" ] && UPLOAD_REPO=$(git remote get-url origin 2>/dev/null | sed 's|.*/||; s|\.git$||') + [ -z "$UPLOAD_REPO" ] && UPLOAD_REPO=$(basename "$(git rev-parse --show-toplevel 2>/dev/null)" 2>/dev/null) + [ -z "$UPLOAD_REPO" ] && UPLOAD_REPO="unknown" + fi + UPLOAD_BRANCH="${RVS_BUILD_REF_NAME:-${GITHUB_REF_NAME:-$(git rev-parse --abbrev-ref HEAD 2>/dev/null || echo "unknown")}}" + UPLOAD_BRANCH=$(echo "$UPLOAD_BRANCH" | sed 's|[^a-zA-Z0-9._/-]|-|g') + UPLOAD_DATE=$(date +%Y-%m-%d) + UPLOAD_SUBPATH="${UPLOAD_REPO}/${UPLOAD_BRANCH}/${UPLOAD_DATE}" + + print_info "Upload path: .../${UPLOAD_SUBPATH}/" + + PKGS=$(find "$BUILD_DIR" -maxdepth 1 -name 'amdrocm*-rvs*' \( -name '*.deb' -o -name '*.rpm' -o -name '*.tar.gz' \) 2>/dev/null) + if [ -z "$PKGS" ]; then + print_error "No packages found to upload" + else + UPLOAD_OK=true + + case "$UPLOAD_TARGET" in + scp://*) + SCP_DEST="${UPLOAD_TARGET#scp://}" + SCP_DEST="${SCP_DEST%/}/${UPLOAD_SUBPATH}/" + print_info "Uploading via SCP to $SCP_DEST" + # Create remote directory first + SCP_HOST="${SCP_DEST%%:*}" + SCP_PATH="${SCP_DEST#*:}" + ssh "$SCP_HOST" "mkdir -p '$SCP_PATH'" 2>/dev/null || true + for pkg in $PKGS; do + print_info " $(basename "$pkg")" + if ! scp "$pkg" "$SCP_DEST"; then + print_error "SCP upload failed for $(basename "$pkg")" + UPLOAD_OK=false + fi + done + ;; + rsync://*) + RSYNC_DEST="${UPLOAD_TARGET#rsync://}" + RSYNC_DEST="${RSYNC_DEST%/}/${UPLOAD_SUBPATH}/" + print_info "Uploading via rsync to $RSYNC_DEST" + if ! echo "$PKGS" | xargs -I{} rsync -avz --progress {} "$RSYNC_DEST"; then + print_error "Rsync upload failed" + UPLOAD_OK=false + fi + ;; + http://*|https://*) + UPLOAD_URL="${UPLOAD_TARGET%/}/${UPLOAD_SUBPATH}" + print_info "Uploading via HTTP PUT to $UPLOAD_URL/" + for pkg in $PKGS; do + local_name=$(basename "$pkg") + print_info " $local_name" + HTTP_CODE=$(curl -s -o /dev/null -w "%{http_code}" \ + -X PUT -T "$pkg" \ + "${UPLOAD_URL}/${local_name}") + if [ "$HTTP_CODE" -ge 200 ] && [ "$HTTP_CODE" -lt 300 ]; then + print_success " Uploaded $local_name (HTTP $HTTP_CODE)" + else + print_error " Failed to upload $local_name (HTTP $HTTP_CODE)" + UPLOAD_OK=false + fi + done + # Show shareable URL (resolve localhost to real IP for team sharing) + if [[ "$UPLOAD_TARGET" =~ localhost|127\.0\.0\.1 ]]; then + SHARE_IP=$(ip -4 route get 1.1.1.1 2>/dev/null | grep -oP 'src \K[0-9.]+' | head -1) + [ -z "$SHARE_IP" ] && SHARE_IP=$(hostname -I 2>/dev/null | awk '{print $1}') + if [ -n "$SHARE_IP" ]; then + SHARE_URL=$(echo "$UPLOAD_URL" | sed "s|localhost|$SHARE_IP|;s|127\.0\.0\.1|$SHARE_IP|") + print_info "Share this URL with your team: ${SHARE_URL}/" + fi + fi + print_info "Browse uploads: ${UPLOAD_URL}/" + ;; + *) + LOCAL_DEST="${UPLOAD_TARGET%/}/${UPLOAD_SUBPATH}" + print_info "Copying packages to $LOCAL_DEST" + mkdir -p "$LOCAL_DEST" + for pkg in $PKGS; do + print_info " $(basename "$pkg")" + if ! cp "$pkg" "$LOCAL_DEST/"; then + print_error "Copy failed for $(basename "$pkg")" + UPLOAD_OK=false + fi + done + ;; + esac + + if [ "$UPLOAD_OK" = true ]; then + print_success "All packages uploaded successfully" + else + print_warning "Some uploads failed - check errors above" + fi + fi + echo "" +fi + +# Summary +echo "================================================================================" +print_success "Package build completed successfully!" +echo "================================================================================" +print_info "Generated packages are in: $BUILD_DIR" +echo "" +# List only package types that were actually generated (DEB on Ubuntu, RPM on CentOS/RHEL, TGZ on all) +PKGS=$(find "$BUILD_DIR" -maxdepth 1 -name 'amdrocm*-rvs*' \( -name '*.deb' -o -name '*.rpm' -o -name '*.tar.gz' \) 2>/dev/null) +if [ -n "$PKGS" ]; then + echo "$PKGS" | xargs ls -lh +else + print_warning "No packages found in build directory" +fi +echo "" + +# Installation instructions +echo "================================================================================" +echo " Installation Instructions" +echo "================================================================================" +echo "" +echo "Ubuntu/Debian (DEB):" +echo " sudo dpkg -i $BUILD_DIR/amdrocm${ROCM_MAJOR}-rvs_*.deb" +echo "" +echo "CentOS/RHEL (RPM):" +echo " sudo rpm -i --replacefiles --nodeps $BUILD_DIR/amdrocm${ROCM_MAJOR}-rvs-*.rpm" +echo "" +echo "Any Linux (TGZ - Relocatable):" +echo " sudo mkdir -p /opt/rocm/extras-${ROCM_MAJOR}" +echo " sudo tar -xzf $BUILD_DIR/amdrocm${ROCM_MAJOR}-rvs-*.tar.gz -C /opt/rocm/extras-${ROCM_MAJOR}" +echo " export PATH=/opt/rocm/extras-${ROCM_MAJOR}/bin:\$PATH" +echo " export LD_LIBRARY_PATH=/opt/rocm/extras-${ROCM_MAJOR}/lib:\$LD_LIBRARY_PATH" +echo "" +echo "================================================================================" + +print_success "Done!" diff --git a/cmake_modules/RVSPackagedRpath.cmake b/cmake_modules/RVSPackagedRpath.cmake new file mode 100644 index 000000000..97bc1e880 --- /dev/null +++ b/cmake_modules/RVSPackagedRpath.cmake @@ -0,0 +1,56 @@ +# Canonical relocatable RUNPATH for packaged RVS artifacts (DEB/RPM/TGZ). +# Used by CMAKE_INSTALL_RPATH and cpack-patch-rpath.cmake (via configure_file). +# +# Expects ROCM_PATH and ROCM_MAJOR_VERSION in the including scope. + +# libomp.so lives under lib/llvm/lib// when ROCm LLVM uses +# LLVM_ENABLE_PER_TARGET_RUNTIME_DIR. +set(RVS_HOST_TARGET_TRIPLE "") +find_program(RVS_ROCM_CLANG + NAMES amdclang++ clang++ + HINTS "${ROCM_PATH}/bin" "${ROCM_PATH}/lib/llvm/bin" "/opt/rocm" "/opt/rocm/core-${ROCM_MAJOR_VERSION}" + PATH_SUFFIXES bin llvm/bin + NO_CACHE) +if(RVS_ROCM_CLANG) + execute_process( + COMMAND "${RVS_ROCM_CLANG}" --print-target-triple + OUTPUT_VARIABLE RVS_HOST_TARGET_TRIPLE + OUTPUT_STRIP_TRAILING_WHITESPACE + RESULT_VARIABLE _rvs_triple_rc + ERROR_QUIET) + if(_rvs_triple_rc EQUAL 0 AND RVS_HOST_TARGET_TRIPLE) + message(STATUS + "RVS LLVM host runtime dir (libomp): /opt/rocm/lib/llvm/lib/${RVS_HOST_TARGET_TRIPLE}") + else() + message(WARNING + "Could not query --print-target-triple from ${RVS_ROCM_CLANG}; " + "libomp may not be found at runtime.") + set(RVS_HOST_TARGET_TRIPLE "") + endif() +else() + message(WARNING + "No ROCm clang found under ${ROCM_PATH}; libomp per-target RPATH omitted.") +endif() + +function(rvs_get_packaged_rpath_list out_var) + set(_rpath + "\$ORIGIN" + "\$ORIGIN/../lib" + "\$ORIGIN/../lib/rvs" + "/opt/rocm/core-${ROCM_MAJOR_VERSION}/lib" + "/opt/rocm/core-${ROCM_MAJOR_VERSION}/lib/llvm/lib" + "/opt/rocm/lib" + "/opt/rocm/lib/llvm/lib") + if(RVS_HOST_TARGET_TRIPLE) + list(APPEND _rpath + "/opt/rocm/core-${ROCM_MAJOR_VERSION}/lib/llvm/lib/${RVS_HOST_TARGET_TRIPLE}" + "/opt/rocm/lib/llvm/lib/${RVS_HOST_TARGET_TRIPLE}") + endif() + set(${out_var} ${_rpath} PARENT_SCOPE) +endfunction() + +function(rvs_get_packaged_rpath_colon out_var) + rvs_get_packaged_rpath_list(_list) + string(JOIN ":" _colon ${_list}) + set(${out_var} "${_colon}" PARENT_SCOPE) +endfunction() diff --git a/cmake_modules/cpack-patch-rpath.cmake.in b/cmake_modules/cpack-patch-rpath.cmake.in new file mode 100644 index 000000000..b83a5273e --- /dev/null +++ b/cmake_modules/cpack-patch-rpath.cmake.in @@ -0,0 +1,48 @@ +# CPack pre-build script: normalize RUNPATH on staged ELF files before DEB/RPM/TGZ. +# Generated at configure time — do not edit. +# Local cmake --build trees are not modified; only CPack staging is patched. + +find_program(PATCHELF_EXECUTABLE patchelf REQUIRED) + +if(NOT DEFINED CPACK_TOPLEVEL_DIRECTORY OR NOT CPACK_TOPLEVEL_DIRECTORY) + message(FATAL_ERROR "cpack-patch-rpath: CPACK_TOPLEVEL_DIRECTORY is not set") +endif() + +set(_rvs_rpath "@RVS_PACKAGED_RPATH_COLON@") +set(_patched 0) +set(_skipped 0) + +file(GLOB_RECURSE _candidates LIST_DIRECTORIES false "${CPACK_TOPLEVEL_DIRECTORY}/*") + +foreach(_candidate IN LISTS _candidates) + if(NOT EXISTS "${_candidate}") + continue() + endif() + # Patch versioned ELF .so files only; skip symlinks (libfoo.so -> libfoo.so.0). + # get_filename_component REALPATH needs CMake 3.20+; symlinks inherit the target RUNPATH. + if(IS_SYMLINK "${_candidate}") + continue() + endif() + if(IS_DIRECTORY "${_candidate}") + continue() + endif() + + file(READ "${_candidate}" _magic OFFSET 0 LIMIT 4 HEX) + if(NOT _magic STREQUAL "7f454c46") + continue() + endif() + + execute_process( + COMMAND "${PATCHELF_EXECUTABLE}" --set-rpath "${_rvs_rpath}" "${_candidate}" + RESULT_VARIABLE _patch_rc + ERROR_VARIABLE _patch_err) + if(_patch_rc EQUAL 0) + math(EXPR _patched "${_patched} + 1") + message(STATUS "cpack-patch-rpath: ${_candidate}") + else() + math(EXPR _skipped "${_skipped} + 1") + message(WARNING "cpack-patch-rpath: failed on ${_candidate}: ${_patch_err}") + endif() +endforeach() + +message(STATUS "cpack-patch-rpath: patched ${_patched} ELF file(s), ${_skipped} failure(s)") diff --git a/cmake_modules/utils.cmake b/cmake_modules/utils.cmake index 4c3dafc21..6511c67e6 100644 --- a/cmake_modules/utils.cmake +++ b/cmake_modules/utils.cmake @@ -118,3 +118,112 @@ function ( get_version DEFAULT_VERSION_STRING ) set( VERSION_BUILD "${VERSION_BUILD}" PARENT_SCOPE ) endfunction() + +## Configure Copyright File for Debian Package +function( configure_pkg PACKAGE_NAME_T COMPONENT_NAME_T PACKAGE_VERSION_T MAINTAINER_NM_T MAINTAINER_EMAIL_T) + # Check If Debian Platform + find_file (DEBIAN debian_version debconf.conf PATHS /etc) + if(DEBIAN) + set( BUILD_DEBIAN_PKGING_FLAG ON CACHE BOOL "Internal Status Flag to indicate Debian Packaging Build" FORCE ) + set_debian_pkg_cmake_flags( ${PACKAGE_NAME_T} ${PACKAGE_VERSION_T} + ${MAINTAINER_NM_T} ${MAINTAINER_EMAIL_T} ) + + # Create debian directory in build tree + file(MAKE_DIRECTORY "${CMAKE_BINARY_DIR}/DEBIAN") + + # Configure the copyright file + configure_file( + "${CMAKE_SOURCE_DIR}/DEBIAN/copyright.in" + "${CMAKE_BINARY_DIR}/DEBIAN/copyright" + @ONLY + ) + + # Install copyright file + install ( FILES "${CMAKE_BINARY_DIR}/DEBIAN/copyright" + DESTINATION "${CMAKE_INSTALL_DOCDIR}" + COMPONENT ${COMPONENT_NAME_T} ) + + # Configure the changelog file + configure_file( + "${CMAKE_SOURCE_DIR}/CHANGELOG.md" + "${CMAKE_BINARY_DIR}/DEBIAN/CHANGELOG.md" + @ONLY + ) + + if( BUILD_ENABLE_LINTIAN_OVERRIDES ) + if(DEFINED BUILD_SHARED_LIBS AND NOT ${BUILD_SHARED_LIBS} STREQUAL "") + string(FIND ${DEB_OVERRIDES_INSTALL_FILENM} "static" OUT_VAR1) + if(OUT_VAR1 EQUAL -1) + set( DEB_OVERRIDES_INSTALL_FILENM "${DEB_OVERRIDES_INSTALL_FILENM}-static" ) + endif() + else() + if(ENABLE_ASAN_PACKAGING) + string( FIND ${DEB_OVERRIDES_INSTALL_FILENM} "asan" OUT_VAR2) + if(OUT_VAR2 EQUAL -1) + set( DEB_OVERRIDES_INSTALL_FILENM "${DEB_OVERRIDES_INSTALL_FILENM}-asan" ) + endif() + endif() + endif() + endif() + + # Install Change Log + find_program ( DEB_GZIP_EXEC gzip ) + if(EXISTS "${CMAKE_BINARY_DIR}/DEBIAN/CHANGELOG.md" ) + execute_process( + COMMAND ${DEB_GZIP_EXEC} -f -n -9 "${CMAKE_BINARY_DIR}/DEBIAN/CHANGELOG.md" + WORKING_DIRECTORY "${CMAKE_BINARY_DIR}/DEBIAN" + RESULT_VARIABLE result + OUTPUT_VARIABLE output + ERROR_VARIABLE error + ) + if(NOT ${result} EQUAL 0) + message(FATAL_ERROR "Failed to compress: ${error}") + endif() + install ( FILES "${CMAKE_BINARY_DIR}/DEBIAN/${DEB_CHANGELOG_INSTALL_FILENM}" + DESTINATION ${CMAKE_INSTALL_DOCDIR} + COMPONENT ${COMPONENT_NAME_T}) + endif() + + else() + # License file + install ( FILES ${LICENSE_FILE} + DESTINATION ${CMAKE_INSTALL_DOCDIR} RENAME LICENSE.txt + COMPONENT ${COMPONENT_NAME_T}) + endif() +endfunction() + +# Set variables for changelog and copyright +# For Debian specific Packages +function( set_debian_pkg_cmake_flags DEB_PACKAGE_NAME_T DEB_PACKAGE_VERSION_T DEB_MAINTAINER_NM_T DEB_MAINTAINER_EMAIL_T ) + # Setting configure flags + set( DEB_PACKAGE_NAME "${DEB_PACKAGE_NAME_T}" CACHE STRING "Debian Package Name" ) + set( DEB_PACKAGE_VERSION "${DEB_PACKAGE_VERSION_T}" CACHE STRING "Debian Package Version String" ) + set( DEB_MAINTAINER_NAME "${DEB_MAINTAINER_NM_T}" CACHE STRING "Debian Package Maintainer Name" ) + set( DEB_MAINTAINER_EMAIL "${DEB_MAINTAINER_EMAIL_T}" CACHE STRING "Debian Package Maintainer Email" ) + set( DEB_COPYRIGHT_YEAR "2025" CACHE STRING "Debian Package Copyright Year" ) + set( DEB_LICENSE "MIT" CACHE STRING "Debian Package License Type" ) + set( DEB_CHANGELOG_INSTALL_FILENM "CHANGELOG.md.gz" CACHE STRING "Debian Package ChangeLog File Name" ) + + if( BUILD_ENABLE_LINTIAN_OVERRIDES ) + set( DEB_OVERRIDES_INSTALL_FILENM "${DEB_PACKAGE_NAME}" CACHE STRING "Debian Package Lintian Override File Name" ) + set( DEB_OVERRIDES_INSTALL_PATH "/usr/share/lintian/overrides/" CACHE STRING "Deb Pkg Lintian Override Install Loc" ) + endif() + + # Get TimeStamp + find_program( DEB_DATE_TIMESTAMP_EXEC date ) + set ( DEB_TIMESTAMP_FORMAT_OPTION "-R" ) + execute_process ( + COMMAND ${DEB_DATE_TIMESTAMP_EXEC} ${DEB_TIMESTAMP_FORMAT_OPTION} + OUTPUT_VARIABLE TIMESTAMP_T + ) + set( DEB_TIMESTAMP "${TIMESTAMP_T}" CACHE STRING "Current Time Stamp for Copyright/Changelog" ) + + message(STATUS "DEB_PACKAGE_NAME : ${DEB_PACKAGE_NAME}" ) + message(STATUS "DEB_PACKAGE_VERSION : ${DEB_PACKAGE_VERSION}" ) + message(STATUS "DEB_MAINTAINER_NAME : ${DEB_MAINTAINER_NAME}" ) + message(STATUS "DEB_MAINTAINER_EMAIL : ${DEB_MAINTAINER_EMAIL}" ) + message(STATUS "DEB_COPYRIGHT_YEAR : ${DEB_COPYRIGHT_YEAR}" ) + message(STATUS "DEB_LICENSE : ${DEB_LICENSE}" ) + message(STATUS "DEB_TIMESTAMP : ${DEB_TIMESTAMP}" ) + message(STATUS "DEB_CHANGELOG_INSTALL_FILENM : ${DEB_CHANGELOG_INSTALL_FILENM}" ) +endfunction() diff --git a/docs/INSTALL_TGZ.md b/docs/INSTALL_TGZ.md new file mode 100644 index 000000000..df1d1f0a8 --- /dev/null +++ b/docs/INSTALL_TGZ.md @@ -0,0 +1,83 @@ +# Installing RVS from the Linux TGZ (`.tar.gz`) archive + +This page describes how to **install and run** the ROCm Validation Suite (RVS) when you have the **relocatable tar.gz** produced by the project’s packaging (e.g. `amdrocm7-rvs-1.3.15-r0711.20260423-Linux.tar.gz`). + +- **Build pipeline, file naming, and S3 download paths** are documented in [README_BUILD_PACKAGES (GitHub workflow)](../.github/workflows/README_BUILD_PACKAGES.md#any-linux-distribution-tgz---relocatable). +- The **same install steps** are mirrored here so the main [README](../README.md) can link to a single, copy-paste-friendly user-facing guide. + +--- + +## Pre-install: ROCm + +Before you install the RVS TGZ, **ROCm must be installed, configured, and on your `PATH` / `LD_LIBRARY_PATH` as in AMD’s documentation** so HIP, HSA, LLVM, and other ROCm libraries are discoverable. Install a stack that includes **ROCm’s LLVM** (for example the **`rocm-llvm`** package, or an equivalent meta-package from your ROCm repo). Follow **rocm docs**: + +- **ROCm documentation (start here)**: +- **Linux install / deployment (paths, env, post-install)**: + +**Assumptions (TGZ use):** ROCm is set up on the machine, `ROCM_PATH` (or your install prefix) is correct, and the runtime can load ROCm libraries. The RVS build records **`RPATH`** entries with **`$ORIGIN`-relative** paths for the extras install, plus **`/opt/rocm/lib`**, **`/opt/rocm/lib/llvm/lib`** (older ROCm layouts), **`/opt/rocm/core-/lib`**, and **`/opt/rocm/core-/lib/llvm/lib`** for the ROCm stack and LLVM (including OpenMP from ROCm’s LLVM). The TGZ only ships RVS; it does not replace a full ROCm stack. + +--- + +## Install RVS run-time dependencies (on the target system) + +Install **PCI** access for GPU enumeration where your distro does not already provide it: + +| Family | Typical package | +|--------|-----------------| +| **Debian / Ubuntu** | `libpci3` (or `libpci-3-0-0` on some releases) | +| **RHEL / Rocky / Alma** | `pciutils-libs` | +| **SUSE / openSUSE** | `libpci3` / `pciutils` as appropriate for the release | + +```bash +# Ubuntu / Debian +sudo apt update && sudo apt install -y libpci3 + +# RHEL / Rocky / Alma 8+ +sudo dnf install -y pciutils-libs + +# openSUSE / SUSE +sudo zypper install libpci3 +``` + +The **DEB/RPM** packages from this project declare a dependency on **`rocm-llvm`** (ROCm LLVM, including the OpenMP runtime used at link time). Prefer installing RVS via those packages when possible so dependencies resolve automatically. + +--- + +## Extract the TGZ + +```bash +# Example for ROCm major 7: match the extras path to your layout and the tarball name. +sudo mkdir -p /opt/rocm/extras-7 +sudo tar -xzf amdrocm7-rvs-*.tar.gz -C /opt/rocm/extras-7 +``` + +--- + +## Post-install: `PATH` and `LD_LIBRARY_PATH` + +Point the shell at the extracted RVS prefix and ROCm (adjust paths to match your install): + +```bash +export ROCM_PATH=/opt/rocm # or your real ROCm root, per rocm docs +export PATH=/opt/rocm/extras-7/bin:$ROCM_PATH/bin:$PATH +export LD_LIBRARY_PATH=/opt/rocm/extras-7/lib:$ROCM_PATH/lib:$ROCM_PATH/lib/llvm/lib:${LD_LIBRARY_PATH} +``` + +**`rvs`** embeds **`RPATH`** for the ROCm stack and LLVM (see **Pre-install: ROCm**), including **`/opt/rocm/lib/llvm/lib`** where applicable, so it usually does not need **`LD_LIBRARY_PATH`** for them. The export still adds **`$ROCM_PATH/lib/llvm/lib`** so your environment matches a typical ROCm tree and works for other tools; if LLVM on your system is only under **`$ROCM_PATH/core-/lib/llvm/lib`**, include that path instead or as well. + +--- + +## Run RVS + +```bash +rvs --help +``` + +(You can also add `/opt/rocm/extras-7/bin` to your user profile or system `PATH` so `rvs` is on the default path in new shells.) + +--- + +## See also + +- [CI / packaging: README_BUILD_PACKAGES.md](../.github/workflows/README_BUILD_PACKAGES.md) — building packages, S3, DEB/RPM, and the full **Installing generated packages** section +- [User guide: module configuration and examples](ug1main.md) diff --git a/docs/cli.md b/docs/cli.md index 2afbc699c..1dfd59a9c 100644 --- a/docs/cli.md +++ b/docs/cli.md @@ -2,7 +2,7 @@ ## Synopsis -rvs [-h|-g|-t|--version|--help|--listTests|--listGpus] +rvs [-h|-g|--version|--help|--listTests|--listGpus] rvs [[[-d|--debugLevel] 0|--quiet] | [[-d|--debugLevel] 1|2|3|4] | [[-d|--debugLevel] 5|--verbose|-v]] [-c path/config_file] [-l path/log_file [-a] [-j]] @@ -16,6 +16,9 @@ -c --config Specify the test configuration file to use. This is a mandatory field for test execution. +-r --run Specify the test level to run. Valid range is 1 to 5, with 5 + indicating the highest stress test level. + -d --debugLevel Specify the debug level for the output log. The range is 0-5 with 5 being the highest verbose level. @@ -24,22 +27,37 @@ -i --indexes Comma separated list of GPU ids/indexes to run test on. This overrides the device/device_index values specified for every actions in the - configuration file, including the ‘all’ value. + configuration file, including the 'all' value. + +-s --selectActions Comma separated list of action names or 0-based action index numbers + to run from the configuration file. Only the matching actions will + be executed. All other actions are skipped. -j --json Generate output file in JSON format. if a path follows this argument, that will be used as json log file; else a file created in /var/tmp/ with timestamp in name. + -l --debugLogFile Generate log file with output and debug information. +-m --module Specify a module name to run the corresponding platform-specific + (MI-series GPUs) module configuration file. Valid modules: babel, + gpup, gst, iet, mem, pebb, peqt, pbqt, rcqt. --t --listTests List the test modules present in RVS. +-t --duration Specify the test duration (in seconds) for each action. Overrides + the duration value in all actions of the configuration file. + + --listTests List the test modules present in RVS. -v --verbose Enable verbose reporting. Equivalent to specifying -d 5 option. +-p --parallel Enables or disables parallel execution across multiple GPUs. Use in + conjunction with -c option. Accepted values: true, false. If no + value is provided, defaults to true. + -n --numTimes Number of times the test repeatedly executes. Use in conjunction with -c option. - --quiet No console output given. See logs and return code for errors. +-q --quiet No console output given. See logs and return code for errors. --version Display version information and exit. diff --git a/docs/conceptual/rvs-modules.rst b/docs/conceptual/rvs-modules.rst index 4a44ce581..97dfa4d0e 100644 --- a/docs/conceptual/rvs-modules.rst +++ b/docs/conceptual/rvs-modules.rst @@ -14,19 +14,6 @@ GPU Properties (GPUP) The GPU Properties module queries the configuration of a target device and returns the device’s static characteristics. These static values can be used to debug issues such as device support, performance and firmware problems. -GPU Monitor (GM module) [deprecated] ------------------------- - -The GPU monitor tool is capable of running on one, some or all of the GPU(s) installed and will report various information at regular intervals. The module can be configured to halt another RVS modules execution if one of the quantities exceeds a specified boundary value. - -PCI Express State Monitor (PESM module) [deprecated] --------------------------------------------- - -The PCIe State Monitor tool is used to actively monitor the PCIe interconnect between the host platform and the GPU. The module will register a “listener” on a target GPU’s PCIe interconnect, and log a message whenever it detects a state change. The PESM will be able to detect the following state changes: - -1. PCIe link speed changes -2. GPU power state changes - ROCm Configuration Qualification Tool (RCQT module) ---------------------------------------------------- @@ -42,11 +29,6 @@ The PCIe Qualification Tool is used to qualify the PCIe bus on which the GPU is 3. PCIe link speed 4. PCIe link width -SBIOS Mapping Qualification Tool (SMQT module) [deprecated] ------------------------------------------------ - -The GPU SBIOS mapping qualification tool is designed to verify that a platform’s SBIOS has satisfied the BAR mapping requirements for VDI and Radeon Instinct products for ROCm support. - P2P Benchmark and Qualification Tool (PBQT module) ---------------------------------------------------- diff --git a/docs/conf.py b/docs/conf.py index 2bbe5fa18..bd7f8d1fb 100644 --- a/docs/conf.py +++ b/docs/conf.py @@ -12,18 +12,41 @@ raise ValueError("VERSION not found!") version_number = match[1] -left_nav_title = f"RVS {version_number} Documentation" +left_nav_title = f"ROCm RVS 1.6 Documentation" + +exclude_patterns = [ + 'conceptual/**', + 'how to/**', + 'schemas/**', +] # for PDF output on Read the Docs project = "RVS Documentation" author = "Advanced Micro Devices, Inc." -copyright = "Copyright (c) 2023-2025 Advanced Micro Devices, Inc. All rights reserved." +copyright = "Copyright (c) 2023-2026 Advanced Micro Devices, Inc. All rights reserved." version = version_number release = version_number html_title = left_nav_title extensions = ["rocm_docs"] html_theme = "rocm_docs_theme" -html_theme_options = {"flavor": "rocm"} +html_theme_options = { + "flavor": "rocm-extras", + "header_title": f"ROCm™ RVS 1.6", + "header_link": f"https://rocm.docs.amd.com/projects/ROCmValidationSuite/en/docs-1.6/", + "version_list_link": f"https://rocm.docs.amd.com/projects/ROCmValidationSuite/en/docs-1.6/versions.html", + "link_main_doc": True, +} + +html_theme_options.update( + { + "repository_url": "https://github.com/ROCm/ROCmValidationSuite", + "path_to_docs": "docs", + "use_repository_button": True, + "use_issues_button": True, + "use_download_button": True, + } +) + external_projects_current_project = "rocmvalidationsuite" -external_toc_path = "./sphinx/_toc.yml" +external_toc_path = "./sphinx/_toc.yml" \ No newline at end of file diff --git a/docs/features.md b/docs/features.md index d461b7a12..499b2f28c 100644 --- a/docs/features.md +++ b/docs/features.md @@ -1,20 +1,11 @@ # ROCmValidationSuite Modules -## GPU Properties – GPUP +## GPU Properties – GPUP module The GPU Properties module queries the configuration of a target device and returns the device’s static characteristics. These static values can be used to debug issues such as device support, performance and firmware problems. -## GPU Monitor – GM module [deprecated] -The GPU monitor tool is capable of running on one, some or all of the GPU(s) installed and will report various information at regular intervals. The module can be configured to halt another RVS modules execution if one of the quantities exceeds a specified boundary value. - -## PCI Express State Monitor – PESM module [deprecated] -The PCIe State Monitor tool is used to actively monitor the PCIe interconnect between the host platform and the GPU. The module will register a “listener” on a target GPU’s PCIe interconnect, and log a message whenever it detects a state change. The PESM will be able to detect the following state changes: - -1. PCIe link speed changes -2. GPU power state changes - -## ROCm Configuration Qualification Tool - RCQT module -The ROCm Configuration Qualification Tool ensures the platform is capable of running ROCm applications and is configured correctly. It checks the installed versions of the ROCm components and the platform configuration of the system. This includes checking the dependencies corresponding to the ROCm meta-packages are installed correctly. +## ROCm Configuration Qualification Tool – RCQT module +The ROCm Configuration Qualification Tool ensures the platform is capable of running ROCm applications and is configured correctly. It checks the installed versions of the ROCm components and the platform configuration of the system. This includes verifying that the dependencies corresponding to the ROCm meta-packages are installed correctly. ## PCI Express Qualification Tool – PEQT module The PCIe Qualification Tool is used to qualify the PCIe bus on which the GPU is connected. The qualification test will be capable of determining the following characteristics of the PCIe bus interconnect to a GPU: @@ -24,23 +15,20 @@ The PCIe Qualification Tool is used to qualify the PCIe bus on which the GPU is 3. PCIe link speed 4. PCIe link width -## SBIOS Mapping Qualification Tool – SMQT module [deprecated] -The GPU SBIOS mapping qualification tool is designed to verify that a platform’s SBIOS has satisfied the BAR mapping requirements for VDI and Radeon Instinct products for ROCm support. - ## P2P Benchmark and Qualification Tool – PBQT module The P2P Benchmark and Qualification Tool is designed to provide the list of all GPUs that support P2P and characterize the P2P links between peers. In addition to testing P2P compatibility, this test will perform a peer-to-peer throughput test between all P2P pairs for performance evaluation. The P2P Benchmark and Qualification Tool will allow users to pick a collection of two or more GPUs to run the test. The user will also be able to select whether or not they want to run the throughput test on each of the pairs. ## PCI Express Bandwidth Benchmark – PEBB module The PCIe Bandwidth Benchmark attempts to saturate the PCIe bus with DMA transfers between system memory and a target GPU card’s memory. The maximum bandwidth obtained is reported to help debug low bandwidth issues. The benchmark should be capable of targeting one, some or all of the GPUs installed in a platform, reporting individual benchmark statistics for each. -## GPU Stress Test - GST module -The GPU Stress Test runs various GEMM computations as workloads to stress the GPU FLOPS performance and check whether it meets the configured target GFLOPS. GEMM workloads shall be configured as either operation type or data type. GEMM based on operation types include SGEMM, DGEMM and HGEMM (Single/Double/Half-precision General Matrix Multiplication) - configured using operation parameter. GEMM based on data types include `fp8`, `i8`, `fp16`, `bf16`, `fp32` and `tf32` (`xf32`) - configured using data type parameter. The duration of the test is configurable, both in terms of time (how long to run) and iterations (how many times to run). +## GPU Stress Test – GST module +The GPU Stress Test runs various GEMM computations as workloads to stress the GPU FLOPS performance and check whether it meets the configured target GFLOPS. GEMM workloads shall be configured as either operation type or data type. GEMM based on operation types include SGEMM, DGEMM and HGEMM (Single/Double/Half-precision General Matrix Multiplication) - configured using operation parameter. GEMM based on data types include `fp8`, `i8`, `fp16`, `bf16`, `fp32` and `tf32` (`xf32`) - configured using data type parameter. The duration of the test is configurable, both in terms of time (how long to run) and iterations (how many times to run). -## Input EDPp Test - IET module +## Input EDPp Test – IET module The Input EDPp Test runs GEMM workloads to stress the GPU power (i.e. TGP). This test is used to verify if the GPU is capable of handling max. power stress for a sustained period of time. Also checks whether GPU power reaches a set target power. -## Memory Test - MEM module -The Memory module tests the GPU memory for hard and soft errors using HIP. It consists of various tests that use algorithms like Walking 1 bit, Moving inversion and Modulo 20. The module executes the following memory tests [Algorithm, data pattern] +## Memory Test – MEM module +The Memory module tests the GPU memory for hard and soft errors using HIP. It consists of various tests that use algorithms like Walking 1 bit, Moving inversion and Modulo 20. The module executes the following memory tests: 1. Walking 1 bit 2. Own address test @@ -53,9 +41,14 @@ The Memory module tests the GPU memory for hard and soft errors using HIP. It co 9. Modulo 20, random pattern 10. Memory stress test -## BABEL benchmark Test - BABEL module +## BABEL Benchmark Test – BABEL module The Babel module executes BabelStream (synthetic GPU benchmark based on the original STREAM benchmark for CPUs) benchmark that measures memory transfer rates (bandwidth) to and from global device memory. Various benchmark tests are implemented using GPU kernels in HIP (Heterogeneous Interface for Portability) programming language. -## Thermal Stress Test - TST module -The Thermal Stress Test (TST) measures/monitors the GPU edge and junction temperatures under various stressful workloads. Also checks whether GPU junction temperature reaches target and trottle temperatures. +## Thermal Stress Test – TST module +The Thermal Stress Test (TST) measures/monitors the GPU edge and junction temperatures under various stressful workloads. Also checks whether GPU junction temperature reaches target and throttle temperatures. + +## Pulse Stress Test – PULSE module +The Pulse Stress Test creates power fluctuations by alternating between high-compute (GEMM) and idle phases at a configurable rate. A two-level barrier synchronizes all GPUs so they spike current simultaneously, maximizing stress on the PSU. + +**PS: Beta version — not intended for production use. Pass/fail criteria are still being tuned.** diff --git a/docs/index.rst b/docs/index.rst index 2a24ebfbc..bf89738e0 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -12,30 +12,29 @@ RVS is a collection of tests, benchmarks, and qualification tools, each targetin For more information, refer to `GitHub. `_ +.. note:: + TransferBench is now a part of the ROCm Validation Suite and is installed with it. + See the `TransferBench documentation `__ for more information. + + .. grid:: 2 :gutter: 3 .. grid-item-card:: Install - * :doc:`ROCm Validation Suite installation <./install/installation>` - + * :doc:`Install ROCm Validation Suite <./install/installation>` - .. grid-item-card:: How to + .. grid-item-card:: Reference - * :doc:`Configure ROCm Validation Suite ` + * :doc:`User guide ` .. grid-item-card:: Conceptual - * :doc:`ROCm Validation Suite modules <./conceptual/rvs-modules>` - + * :doc:`ROCm Validation Suite modules ` + * :doc:`TransferBench in RVS ` To contribute to the documentation, refer to `Contributing to ROCm `_. You can find licensing information on the `Licensing `_ page. - - - - - diff --git a/docs/install/installation.rst b/docs/install/installation.rst index 65989d85d..f01d37e6f 100644 --- a/docs/install/installation.rst +++ b/docs/install/installation.rst @@ -1,293 +1,615 @@ .. meta:: - :description: Install ROCm Validation Suite - :keywords: install, rocm validation suite, rvs, RVS, AMD, ROCm + :description lang=en: Install ROCm Validation Suite (RVS) + :keywords: rocm, core, sdk, rvs, validation, suite, install +***************************** +Install ROCm Validation Suite +***************************** -********************************** -Installing ROCm Validation Suite -********************************** - -You can obtain ROCm Validation Suite (RVS) by building it from: +ROCm Validation Suite (RVS) is supported on AMD Instinct and Radeon GPUs +supported by ROCm. See the `ROCm compatibility matrix `__ for support information. -* the source code base +For advanced workflows, source builds, or custom configurations, see +``__. -* a prebuilt package +.. note:: + TransferBench is now a part of the ROCm Validation Suite and is installed with it. + See the `TransferBench documentation `__ for more information. -Building from source code ---------------------------- +Prerequisites +============= + +Install the ROCm Core SDK before installing RVS. + +For instructions, see `Install AMD ROCm `__. Use the +selector panel on that page to view instructions appropriate for your system environment. + +ROCm installation path +---------------------- + +The ROCm installation path depends on what installation method was selected during +ROCm installation. After installation, the ROCm Core SDK will be deployed to this +path location and contains all the core ROCm directories such as ``bin``, ``include``, +and ``lib``. When configuring the environment for RVS, you must set the +``ROCM_INSTALL_PATH`` environment variable to the ROCm core installation directory +```` based on the installation method used: + +.. tab-set:: + + .. tab-item:: Package manager + + .. code-block:: bash + + ROCM_INSTALL_PATH=/opt/rocm/core-10.1 # = /opt/rocm/core-10.1 -RVS is an open-source solution. For more details, refer to the `ROCm Validation Suite GitHub repository. `_ + .. tab-item:: pip + Use within active python virtual environment: + + .. code-block:: bash + + ROCM_INSTALL_PATH=$(rocm-sdk path --root) # = rocm-sdk path of core installation + + .. tab-item:: Tarball + + .. code-block:: bash + + ROCM_INSTALL_PATH=/therock-tarball/install # = to default "therock-tarball" installation + + .. tab-item:: Runfile + + .. code-block:: bash + + ROCM_INSTALL_PATH=/opt/rocm/core-10.1 # = default /opt/rocm/core-10.1 installation (no target=) + ROCM_INSTALL_PATH=/rocm/core-10.1 # = target= Package manager installation ------------------------------- - -Based on the OS, use the appropriate package manager to install the RVS package. +============================= -For more details, refer to the `ROCm Validation Suite GitHub repository. `_ +Use the following steps to install RVS using your distribution's package manager +on top of the ROCm Core SDK. -RVS package components are installed in ``/opt/rocm``. The package contains: +1. Register the RVS repository. -- executable binary, located in ``_install-base_/bin/rvs``. -- public shared libraries, located in ``_install-base_/lib``. -- module specific shared libraries, located in ``_install-base_/lib/rvs``. -- default configuration files, located in ``_install-base_/share/rocm-validation-suite/conf``. -- GPU specific configuration files, located in ``_install-base_/share/rocm-validation-suite/conf/``. -- testscripts, located in ``_install-base_/share/rocm-validation-suite/testscripts``. -- user guide, located in ``_install-base_/share/rocm-validation-suite/userguide``. -- man page, located in ``_install-base_/share/man``. + .. tab-set:: + + .. tab-item:: Ubuntu + :sync: ubuntu + + .. tab-set:: + + .. tab-item:: 26.04 + :sync: ubuntu-2604 + + .. code-block:: bash + + sudo mkdir --parents --mode=0755 /etc/apt/keyrings + wget https://stable.repo.amd.com/rocm/gpg/packages.gpg -O - | gpg --dearmor | sudo tee /etc/apt/keyrings/amdrocm.gpg > /dev/null + sudo tee /etc/apt/sources.list.d/amdrocm-rvs.sources << 'EOF' + X-Repo-Id: amdrocm-rvs + Types: deb + URIs: https://stable.repo.amd.com/rocm/extras/rvs/packages/ubuntu2604/ + Suites: stable + Components: main + Architectures: amd64 + Signed-By: /etc/apt/keyrings/amdrocm.gpg + Enabled: yes + EOF + + sudo apt update + + .. tab-item:: 24.04 + :sync: ubuntu-2404 + + .. code-block:: bash + + sudo mkdir --parents --mode=0755 /etc/apt/keyrings + wget https://stable.repo.amd.com/rocm/gpg/packages.gpg -O - | gpg --dearmor | sudo tee /etc/apt/keyrings/amdrocm.gpg > /dev/null + sudo tee /etc/apt/sources.list.d/amdrocm-rvs.sources << 'EOF' + X-Repo-Id: amdrocm-rvs + Types: deb + URIs: https://stable.repo.amd.com/rocm/extras/rvs/packages/ubuntu2404/ + Suites: stable + Components: main + Architectures: amd64 + Signed-By: /etc/apt/keyrings/amdrocm.gpg + Enabled: yes + EOF + + sudo apt update + + .. tab-item:: 22.04 + :sync: ubuntu-2204 + + .. code-block:: bash + + sudo mkdir --parents --mode=0755 /etc/apt/keyrings + wget https://stable.repo.amd.com/rocm/gpg/packages.gpg -O - | gpg --dearmor | sudo tee /etc/apt/keyrings/amdrocm.gpg > /dev/null + sudo tee /etc/apt/sources.list.d/amdrocm-rvs.sources << 'EOF' + X-Repo-Id: amdrocm-rvs + Types: deb + URIs: https://stable.repo.amd.com/rocm/extras/rvs/packages/ubuntu2204/ + Suites: stable + Components: main + Architectures: amd64 + Signed-By: /etc/apt/keyrings/amdrocm.gpg + Enabled: yes + EOF + + sudo apt update + + .. tab-item:: Debian + :sync: debian + + .. tab-set:: + + .. tab-item:: 13 + :sync: debian-13 + + .. code-block:: bash + + sudo mkdir --parents --mode=0755 /etc/apt/keyrings + wget https://stable.repo.amd.com/rocm/gpg/packages.gpg -O - | gpg --dearmor | sudo tee /etc/apt/keyrings/amdrocm.gpg > /dev/null + sudo tee /etc/apt/sources.list.d/amdrocm-rvs.sources << 'EOF' + X-Repo-Id: amdrocm-rvs + Types: deb + URIs: https://stable.repo.amd.com/rocm/extras/rvs/packages/debian13/ + Suites: stable + Components: main + Architectures: amd64 + Signed-By: /etc/apt/keyrings/amdrocm.gpg + Enabled: yes + EOF -Prerequisites ------------------- + sudo apt update -RVS has been tested on all ROCm-supported Linux environments except for RHEL 9.4. See `Supported operating systems `_ for the complete list of ROCm-supported Linux environments. + .. tab-item:: 12 + :sync: debian-12 -.. Note:: + .. code-block:: bash - This topic provides commands for the primary Linux distribution families. These commands are also applicable to other operating systems derived from the same families. + sudo mkdir --parents --mode=0755 /etc/apt/keyrings + wget https://stable.repo.amd.com/rocm/gpg/packages.gpg -O - | gpg --dearmor | sudo tee /etc/apt/keyrings/amdrocm.gpg > /dev/null + sudo tee /etc/apt/sources.list.d/amdrocm-rvs.sources << 'EOF' + X-Repo-Id: amdrocm-rvs + Types: deb + URIs: https://stable.repo.amd.com/rocm/extras/rvs/packages/debian12/ + Suites: stable + Components: main + Architectures: amd64 + Signed-By: /etc/apt/keyrings/amdrocm.gpg + Enabled: yes + EOF -Ensure you review the following prerequisites carefully for each operating system before compiling or installing the RVS package. + sudo apt update -.. tab-set:: - .. tab-item:: Ubuntu - :sync: Ubuntu + .. tab-item:: RHEL + :sync: rhel - .. code-block:: shell + .. tab-set:: - sudo apt-get -y update && sudo apt-get install -y libpci3 libpci-dev doxygen unzip cmake git libyaml-cpp-dev + .. tab-item:: 10 + :sync: rhel-10 - .. tab-item:: RHEL - :sync: RHEL - - .. code-block:: shell - - sudo yum install -y cmake3 doxygen rpm rpm-build git gcc-c++ yaml-cpp-devel - - wget http://mirror.centos.org/centos/7/os/x86_64/Packages/pciutils-devel-3.5.1-3.el7.x86_64.rpm - - sudo rpm -ivh pciutils-devel-3.5.1-3.el7.x86_64.rpm + .. code-block:: bash - .. tab-item:: SUSE - :sync: SUSE - - .. code-block:: shell - - sudo zypper install -y cmake doxygen pciutils-devel libpci3 rpm git rpm-build gcc-c++ yaml-cpp-devel + sudo tee /etc/yum.repos.d/amdrocm-rvs.repo <`_ for more details. + sudo tee /etc/yum.repos.d/amdrocm-rvs.repo <``: - .. code-block:: shell + .. code-block:: bash - sudo rpm -e rocm-smi-lib && sudo yum install rocm-smi-lib + ROCM_INSTALL_PATH= # ie. /opt/rocm/core-10.1 - .. tab-item:: SUSE - :sync: SUSE + b. Configure your environment. - .. code-block:: shell + .. tab-set:: - sudo rpm -e rocm-smi-lib && sudo zypper install rocm-smi-lib + .. tab-item:: System-wide + .. code-block:: bash -Building from source ---------------------- + sudo tee /etc/profile.d/set-rvs-env.sh << EOF + export EXTRAS_PATH=/opt/rocm/extras-10 + export ROCM_PATH=$ROCM_INSTALL_PATH + export PATH=\$EXTRAS_PATH/bin:\$ROCM_PATH/bin:\$PATH + export LD_LIBRARY_PATH=\$EXTRAS_PATH/lib:\$ROCM_PATH/lib:\$ROCM_PATH/lib/llvm/lib:\$LD_LIBRARY_PATH + EOF -This section explains how to get and compile the current development stream of RVS. + sudo chmod +x /etc/profile.d/set-rvs-env.sh + source /etc/profile.d/set-rvs-env.sh -1. Clone the repository. + .. tab-item:: User -.. code-block:: + .. code-block:: bash - git clone https://github.com/ROCm/ROCmValidationSuite.git + tee --append ~/.bashrc << EOF + # BEGIN RVS environment configuration + export EXTRAS_PATH=/opt/rocm/extras-10 + export ROCM_PATH=$ROCM_INSTALL_PATH + export PATH=\$EXTRAS_PATH/bin:\$ROCM_PATH/bin:\$PATH + export LD_LIBRARY_PATH=\$EXTRAS_PATH/lib:\$ROCM_PATH/lib:\$ROCM_PATH/lib/llvm/lib:\$LD_LIBRARY_PATH + # END RVS environment configuration + EOF -2. Configure the build system for RVS. + source ~/.bashrc -.. code-block:: +4. Verify your installation. - cd ROCmValidationSuite - cmake -B ./build -DROCM_PATH= -DCMAKE_INSTALL_PREFIX= -DCPACK_PACKAGING_INSTALL_PREFIX= + .. code-block:: bash -For example, if ROCm 5.5 was installed, run the following command: + rvs -g -.. code-block:: +.. note:: + The ROCm repositories must be set up before installing RVS. This repository + setup is part of the ROCm Core SDK installation. - cmake -B ./build -DROCM_PATH=/opt/rocm-5.5.0 -DCMAKE_INSTALL_PREFIX=/opt/rocm-5.5.0 -DCPACK_PACKAGING_INSTALL_PREFIX=/opt/rocm-5.5.0 +Package manager uninstalling +============================ -3. Build the binary. +1. Use your package manager to remove the installed packages. -.. code-block:: + .. tab-set:: - make -C ./build + .. tab-item:: Ubuntu + :sync: ubuntu -4. Build the package. + .. code-block:: bash -.. code-block:: + sudo apt autoremove amdrocm10-rvs - cd ./build - make package + .. tab-item:: RHEL + :sync: rhel -.. Note:: + .. code-block:: bash - Depending on your OS, only DEB or RPM package will be built. + sudo dnf remove amdrocm10-rvs -.. Note:: +2. Remove RVS repositories. - You can ignore errors about unrelated configurations. + .. tab-set:: -5. Install the built package. + .. tab-item:: Ubuntu + :sync: ubuntu -.. tab-set:: - .. tab-item:: Ubuntu - :sync: Ubuntu + .. tab-set:: - .. code-block:: + .. tab-item:: 26.04 + :sync: ubuntu-2604 - sudo dpkg -i rocm-validation-suite*.deb + .. code-block:: bash - .. tab-item:: RHEL - :sync: RHEL + # Remove RVS repositories + sudo rm /etc/apt/sources.list.d/amdrocm-rvs.sources - .. code-block:: shell + # Clear the cache and clean the system + sudo rm -rf /var/cache/apt/* + sudo apt clean all + sudo apt update - sudo rpm -i --replacefiles --nodeps rocm-validation-suite*.rpm + .. tab-item:: 24.04 + :sync: ubuntu-2404 - .. tab-item:: SUSE - :sync: SUSE + .. code-block:: bash - .. code-block:: shell + # Remove RVS repositories + sudo rm /etc/apt/sources.list.d/amdrocm-rvs.sources - sudo rpm -i --replacefiles --nodeps rocm-validation-suite*.rpm + # Clear the cache and clean the system + sudo rm -rf /var/cache/apt/* + sudo apt clean all + sudo apt update -.. Note:: + .. tab-item:: 22.04 + :sync: ubuntu-2204 - RVS is packaged as part of the ROCm release starting from 3.0. You can install the pre-compiled package as indicated below. Ensure prerequisites, ROCm stack, rocblas and rocm-smi-lib64 are already installed. + .. code-block:: bash -6. Install the package included with the ROCm release. + # Remove RVS repositories + sudo rm /etc/apt/sources.list.d/amdrocm-rvs.sources -.. tab-set:: - .. tab-item:: Ubuntu - :sync: Ubuntu + # Clear the cache and clean the system + sudo rm -rf /var/cache/apt/* + sudo apt clean all + sudo apt update + + .. tab-item:: Debian + :sync: debian + + .. tab-set:: + + .. tab-item:: 13 + :sync: debian-13 + + .. code-block:: bash + + # Remove RVS repositories + sudo rm /etc/apt/sources.list.d/amdrocm-rvs.sources + + # Clear the cache and clean the system + sudo rm -rf /var/cache/apt/* + sudo apt clean all + sudo apt update + + .. tab-item:: 12 + :sync: debian-12 + + .. code-block:: bash + + # Remove RVS repositories + sudo rm /etc/apt/sources.list.d/amdrocm-rvs.sources + + # Clear the cache and clean the system + sudo rm -rf /var/cache/apt/* + sudo apt clean all + sudo apt update + + .. tab-item:: RHEL + :sync: rhel + + .. tab-set:: + + .. tab-item:: 10 + :sync: rhel-10 + + .. code-block:: bash + + # Remove RVS repositories + sudo rm /etc/yum.repos.d/amdrocm-rvs.repo + + # Clear the cache and clean the system + sudo rm -rf /var/cache/dnf + sudo dnf clean all + + .. tab-item:: 9 + :sync: rhel-9 + + .. code-block:: bash + + # Remove RVS repositories + sudo rm /etc/yum.repos.d/amdrocm-rvs.repo + + # Clear the cache and clean the system + sudo rm -rf /var/cache/dnf + sudo dnf clean all + + .. tab-item:: 8 + :sync: rhel-8 + + .. code-block:: bash + + # Remove RVS repositories + sudo rm /etc/yum.repos.d/amdrocm-rvs.repo + + # Clear the cache and clean the system + sudo rm -rf /var/cache/dnf + sudo dnf clean all + +3. Remove your RVS environment configuration from your system. + + .. tab-set:: + + .. tab-item:: System-wide + + If you opted for a system-wide setup during the installation + process, remove the RVS environment variables. + + .. code-block:: bash + + sudo rm -f /etc/profile.d/set-rvs-env.sh + + .. tab-item:: User + + If you opted for a user-specific setup during the installation + process, remove the RVS environment configuration block from + your shell configuration file (``~/.bashrc`` or ``~/.profile``). + +Tarball installation +==================== + +Use the following steps to install RVS using a tarball on top of the ROCm Core SDK. + +1. Install system dependencies. + + .. tab-set:: + + .. tab-item:: Ubuntu + :sync: ubuntu + + .. code-block:: bash + + sudo apt install libpci3 libnuma1 + + .. tab-item:: RHEL + :sync: rhel + + .. code-block:: bash + + sudo dnf install pciutils-libs numactl-libs + +2. Download the RVS tarball. + + .. code-block:: bash + + wget https://stable.repo.amd.com/rocm/extras/rvs/tarball/amdrocm10-rvs-1.6.131-844-Linux.tar.gz + + The tarball file may be verified against the following SHA256: + + .. code-block:: bash + + wget https://stable.repo.amd.com/rocm/extras/rvs/tarball/amdrocm10-rvs-1.6.131-844-Linux.tar.gz.sha256 + +3. Extract the tarball to the ROCm Extras location. + + RVS is part of the ROCm Extras set of tools that work with the ROCm Core SDK. + The ROCm Extras location (``EXTRAS_INSTALL_PATH``) can be set to a custom + location ````, but is typically set based on the ROCm installation + method used. + + .. code-block:: bash + + EXTRAS_INSTALL_PATH= # ie. = path to ROCm extras for RVS extract - .. code-block:: + sudo mkdir -p $EXTRAS_INSTALL_PATH + sudo tar -xzf amdrocm10-rvs-1.6.131-844-Linux.tar.gz -C $EXTRAS_INSTALL_PATH - sudo apt install rocm-validation-suite + **Recommended:** Set ``EXTRAS_INSTALL_PATH`` to a location within the root + install directory for the ROCm Core SDK. + For example, if you installed the ROCm Core SDK using your Linux + distribution's package manager: - .. tab-item:: RHEL - :sync: RHEL + .. code-block:: bash - .. code-block:: shell + sudo mkdir -p /opt/rocm/extras-10 + sudo tar -xzf amdrocm10-rvs-1.6.131-844-Linux.tar.gz -C /opt/rocm/extras-10 - sudo yum install rocm-validation-suite +4. Complete the following post-installation steps. - .. tab-item:: SUSE - :sync: SUSE + Use the following commands to update your shell configuration file + (``~/.bashrc`` or ``~/.profile``) and add Extras and ROCm to your PATH. - .. code-block:: shell + a. Set the ROCm installation path based on the ROCm installation method + and ````: - sudo zypper install rocm-validation-suite + .. code-block:: bash + ROCM_INSTALL_PATH= # ie. /opt/rocm/core-10.1 -Reporting ------------ + b. Configure your environment. -Test results, errors, and verbose logs are printed as terminal output. To enable JSON logging, use the ``-j`` option. The JSON output file is stored in the ``/var/tmp`` folder and the file name will be printed. + .. tab-set:: -You can build RVS from the source code base or by installing from a pre-built package. See the preceding sections for more details. + .. tab-item:: System-wide -Running RVS ------------- + .. code-block:: bash -Run the version built from source code -++++++++++++++++++++++++++++++++++++++ + sudo tee /etc/profile.d/set-rvs-env.sh << EOF + export EXTRAS_PATH=$EXTRAS_INSTALL_PATH + export ROCM_PATH=$ROCM_INSTALL_PATH + export PATH=\$EXTRAS_PATH/bin:\$ROCM_PATH/bin:\$PATH + export LD_LIBRARY_PATH=\$EXTRAS_PATH/lib:\$ROCM_PATH/lib:\$ROCM_PATH/lib/llvm/lib:\$LD_LIBRARY_PATH + EOF -.. code-block:: + sudo chmod +x /etc/profile.d/set-rvs-env.sh + source /etc/profile.d/set-rvs-env.sh - cd /build/bin + .. tab-item:: User - Command examples - ./rvs --help ; Lists all options to run RVS test suite - ./rvs -g ; Lists supported GPUs available in the machine - ./rvs -d 3 ; Run set of RVS default sanity tests (in rvs.conf) with verbose level 3 - ./rvs -c conf/gst_single.conf ; Run GST module default test configuration + .. code-block:: bash -Run the version pre-compiled and packaged with the ROCm release -+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ + tee --append ~/.bashrc << EOF + # BEGIN RVS environment configuration + export EXTRAS_PATH=$EXTRAS_INSTALL_PATH + export ROCM_PATH=$ROCM_INSTALL_PATH + export PATH=\$EXTRAS_PATH/bin:\$ROCM_PATH/bin:\$PATH + export LD_LIBRARY_PATH=\$EXTRAS_PATH/lib:\$ROCM_PATH/lib:\$ROCM_PATH/lib/llvm/lib:\$LD_LIBRARY_PATH + # END RVS environment configuration + EOF -.. code-block:: + source ~/.bashrc - cd /opt/rocm/bin +5. Verify your installation. - Command examples - ./rvs --help ; Lists all options to run RVS test suite - ./rvs -g ; Lists supported GPUs available in the machine - ./rvs -d 3 ; Run set of RVS sanity tests (in rvs.conf) with verbose level 3 - ./rvs -c ../share/rocm-validation-suite/conf/gst_single.conf ; Run GST default test configuration + .. code-block:: bash -To run GPU-specific test configurations, use the configuration files in the GPU folders under ``/opt/rocm/share/rocm-validation-suite/conf``. + rvs -g -.. code-block:: +Tarball uninstalling +==================== - ./rvs -c ../share/rocm-validation-suite/conf/MI300X/gst_single.conf ; Run MI300X specific GST test configuration - ./rvs -c ../share/rocm-validation-suite/conf/nv32/gst_single.conf ; Run Navi 32 specific GST test configuration +1. Remove the installation directory. -.. Note:: + .. important:: + The following command assumes you're working with the + ``EXTRAS_INSTALL_PATH`` directory set to ``/opt/rocm/extras-10``. If you + chose a different directory name when installing RVS, adjust the command + accordingly. - Always use GPU-specific configurations over the default test configurations. + .. code-block:: bash -Building documentation ------------------------- + sudo rm -rf /opt/rocm/extras-10 -Run the following commands to build documentation locally. +2. Remove the RVS environment configuration from your system. -.. code-block:: + .. tab-set:: - cd docs - pip3 install -r .sphinx/requirements.txt - python3 -m sphinx -T -E -b html -d _build/doctrees -D language=en . _build/html + .. tab-item:: System-wide + If you opted for a system-wide setup during the installation process, + remove the RVS environment variables. + .. code-block:: bash + sudo rm -f /etc/profile.d/set-rvs-env.sh + .. tab-item:: User + If you opted for a user-specific setup during the installation + process, remove the RVS environment configuration block from your + shell configuration file (``~/.bashrc`` or ``~/.profile``). \ No newline at end of file diff --git a/docs/install/prerequisites.md b/docs/install/prerequisites.md deleted file mode 100644 index f6dbd2cf0..000000000 --- a/docs/install/prerequisites.md +++ /dev/null @@ -1,116 +0,0 @@ -## Prerequisites - -Follow the instructions below before compilation/installing compiled package. - -### Ubuntu: - - sudo apt-get -y update && sudo apt-get install -y libpci3 libpci-dev doxygen unzip cmake git libyaml-cpp-dev - -### CentOS: - - sudo yum install -y cmake3 doxygen pciutils-devel rpm rpm-build git gcc-c++ yaml-cpp-devel - -### RHEL: - - sudo yum install -y cmake3 doxygen rpm rpm-build git gcc-c++ yaml-cpp-devel pciutils-devel - -### SLES: - - sudo zypper install -y cmake doxygen pciutils-devel libpci3 rpm git rpm-build gcc-c++ yaml-cpp-devel - -## Install ROCm stack, rocblas and rocm-smi-lib -Install ROCm stack for Ubuntu/CentOS/SLES/RHEL. Refer to - [ROCm installation guide](https://rocmdocs.amd.com/en/latest/Installation_Guide/Installation-Guide.html) for more details. - -**Note** - -rocm_smi64 package has been renamed to rocm-smi-lib64 from >= ROCm3.0. If you are using ROCm release < 3.0 , install the package as "rocm_smi64". -rocm-smi-lib64 package has been renamed to rocm-smi-lib from >= ROCm4.1. - -Install rocBLAS and rocm-smi-lib : - -### Ubuntu: - - sudo apt-get install rocblas rocm-smi-lib - -### CentOS & RHEL: - - sudo yum install --nogpgcheck rocblas rocm-smi-lib - -### SUSE: - - sudo zypper install rocblas rocm-smi-lib - -**Note** - -If rocm-smi-lib is already installed but /opt/rocm/lib/librocm_smi64.so doesn't exist. Do below: - -### Ubuntu: - - sudo dpkg -r rocm-smi-lib && sudo apt install rocm-smi-lib - -### CentOS & RHEL: - - sudo rpm -e rocm-smi-lib && sudo yum install rocm-smi-lib - -### SUSE: - - sudo rpm -e rocm-smi-lib && sudo zypper install rocm-smi-lib - -## Building from Source -This section explains how to get and compile current development stream of RVS. - -### Clone repository - - git clone https://github.com/ROCm/ROCmValidationSuite.git - -### Configure: - - cd ROCmValidationSuite - cmake -B ./build -DROCM_PATH= -DCMAKE_INSTALL_PREFIX= -DCPACK_PACKAGING_INSTALL_PREFIX= - - e.g. If ROCm 5.5 was installed, - cmake -B ./build -DROCM_PATH=/opt/rocm-5.5.0 -DCMAKE_INSTALL_PREFIX=/opt/rocm-5.5.0 -DCPACK_PACKAGING_INSTALL_PREFIX=/opt/rocm-5.5.0 - -### Build binary: - - make -C ./build - -### Build package: - - cd ./build - make package - -**Note** - -Based on your OS, only DEB or RPM package will be built. You may ignore an error for the unrelated configuration. - -### Install built package: - -#### Ubuntu: - - sudo dpkg -i rocm-validation-suite*.deb - -#### CentOS, RHEL, and SUSE: - - sudo rpm -i --replacefiles --nodeps rocm-validation-suite*.rpm - -**Note** - -RVS is getting packaged as part of ROCm release starting from 3.0. You can install pre-compiled package as below. -Please make sure Prerequisites, ROCm stack, rocblas and rocm-smi-lib64 are already installed - -### Install package packaged with ROCm release: - -#### Ubuntu: - - sudo apt install rocm-validation-suite - -#### CentOS & RHEL: - - sudo yum install rocm-validation-suite - -#### SUSE: - - sudo zypper install rocm-validation-suite - diff --git a/docs/schemas/babel.schema b/docs/schemas/babel.schema index ada3ea986..7e83d71f4 100644 --- a/docs/schemas/babel.schema +++ b/docs/schemas/babel.schema @@ -19,40 +19,78 @@ "gpu_index": { "type": "string" }, - "Array size": { + "array_size": { "type": "string" }, - "Total size": { + "total_size": { "type": "string" }, - "Iterations": { + "duration_ms": { "type": "string" }, - "Function": { + "iterations": { "type": "string" }, - "MBytes/sec": { - "type": "string" - }, - "Min(s)": { - "type": "string" - }, - "Max(s)": { - "type": "string" - }, - "Average(s)": { - "type": "string" - }, - "pass": { - "type": "string" + "results": { + "type": "array", + "items": { + "type": "object", + "properties": { + "subtest": { + "type": "string", + "enum": ["Read", "Write", "Copy", "Mul", "Add", "Triad", "Dot"] + }, + "mbytes_per_sec": { + "type": "string" + }, + "max_mbytes_per_sec": { + "type": "string" + }, + "min_mbytes_per_sec": { + "type": "string" + }, + "avg_mbytes_per_sec": { + "type": "string" + }, + "mibytes_per_sec": { + "type": "string" + }, + "max_mibytes_per_sec": { + "type": "string" + }, + "min_mibytes_per_sec": { + "type": "string" + }, + "avg_mibytes_per_sec": { + "type": "string" + }, + "pass": { + "type": "string" + } + }, + "required": ["subtest", "pass"], + "oneOf": [ + { + "required": ["mbytes_per_sec", "max_mbytes_per_sec", + "min_mbytes_per_sec", "avg_mbytes_per_sec"] + }, + { + "required": ["mibytes_per_sec", "max_mibytes_per_sec", + "min_mibytes_per_sec", "avg_mibytes_per_sec"] + } + ] + }, + "minItems": 1 } - } + }, + "required": ["gpu_id", "gpu_index", "array_size", "total_size", "results"] }, "minItems": 1 } } }, "required": [ + "version", "babel" ] } diff --git a/docs/schemas/tst.schema b/docs/schemas/tst.schema index 916bcdc44..cad336c67 100644 --- a/docs/schemas/tst.schema +++ b/docs/schemas/tst.schema @@ -16,6 +16,9 @@ "target_temp": { "type": "string" }, + "throttle_temp": { + "type": "string" + }, "dtype": { "type": "string" }, @@ -28,6 +31,9 @@ "average edge temperature": { "type": "string" }, + "average junction temperature": { + "type": "string" + }, "pass": { "type": "string" } diff --git a/docs/sphinx/_toc.yml.in b/docs/sphinx/_toc.yml.in index 10656ed63..90ff22d89 100644 --- a/docs/sphinx/_toc.yml.in +++ b/docs/sphinx/_toc.yml.in @@ -2,23 +2,18 @@ defaults: numbered: False root: index subtrees: -- caption: Install - entries: - - file: install/installation.rst - title: ROCm Validation Suite installation - - -- caption: How to - entries: - - file: how to/configure-rvs.rst - title: Configure ROCm Validation Suite - -- caption: Conceptual - entries: - - file: conceptual/rvs-modules.rst - title: ROCm Validation Suite modules - - -- caption: About - entries: - - file: license.md + - caption: Install + entries: + - file: install/installation.rst + title: Install ROCm Validation Suite + - caption: Reference + entries: + - file: ug1main.md + title: User guide + - file: features.md + title: ROCm Validation Suite modules + - file: transferbench.md + title: TransferBench in RVS + - caption: About + entries: + - file: license.md diff --git a/docs/sphinx/requirements.in b/docs/sphinx/requirements.in index e4ada44c2..e30b1f824 100644 --- a/docs/sphinx/requirements.in +++ b/docs/sphinx/requirements.in @@ -1 +1 @@ -rocm-docs-core==1.18.2 +rocm-docs-core @ git+https://github.com/ROCm/rocm-docs-core.git@develop diff --git a/docs/sphinx/requirements.txt b/docs/sphinx/requirements.txt index 46db293d8..fa1daa257 100644 --- a/docs/sphinx/requirements.txt +++ b/docs/sphinx/requirements.txt @@ -1,101 +1,94 @@ -# -# This file is autogenerated by pip-compile with Python 3.10 -# by the following command: -# -# pip-compile requirements.in -# +# This file was autogenerated by uv via the following command: +# uv pip compile /home/pmoutsia/ROCmValidationSuite/docs/sphinx/requirements.in --python-version 3.10 --output-file /home/pmoutsia/ROCmValidationSuite/docs/sphinx/requirements.txt accessible-pygments==0.0.5 # via pydata-sphinx-theme -alabaster==0.7.16 +alabaster==1.0.0 # via sphinx -asttokens==3.0.0 +asttokens==3.0.1 # via stack-data -attrs==25.1.0 +attrs==26.1.0 # via # jsonschema # jupyter-cache # referencing -babel==2.15.0 +babel==2.18.0 # via # pydata-sphinx-theme # sphinx -beautifulsoup4==4.12.3 +beautifulsoup4==4.15.0 # via pydata-sphinx-theme -breathe==4.35.0 +breathe==4.36.0 # via rocm-docs-core -certifi==2024.7.4 +certifi==2026.5.20 # via requests -cffi==1.16.0 +cffi==2.0.0 # via # cryptography # pynacl -charset-normalizer==3.3.2 +charset-normalizer==3.4.7 # via requests -click==8.1.7 +click==8.4.1 # via # jupyter-cache # sphinx-external-toc -comm==0.2.2 +comm==0.2.3 # via ipykernel -cryptography==42.0.8 +cryptography==48.0.1 # via pyjwt -debugpy==1.8.12 +debugpy==1.8.21 # via ipykernel -decorator==5.1.1 +decorator==5.3.1 # via ipython -deprecated==1.2.14 - # via pygithub docutils==0.21.2 # via - # breathe # myst-parser # pydata-sphinx-theme # sphinx -exceptiongroup==1.2.2 +exceptiongroup==1.3.1 # via ipython -executing==2.2.0 +executing==2.2.1 # via stack-data -fastjsonschema==2.20.0 +fastjsonschema==2.21.2 # via # nbformat # rocm-docs-core -gitdb==4.0.11 +gitdb==4.0.12 # via gitpython -gitpython==3.1.43 +gitpython==3.1.50 # via rocm-docs-core -greenlet==3.1.1 +greenlet==3.5.1 # via sqlalchemy -idna==3.7 +idna==3.18 # via requests -imagesize==1.4.1 +imagesize==2.0.0 # via sphinx -importlib-metadata==8.6.1 +importlib-metadata==9.0.0 # via # jupyter-cache # myst-nb -ipykernel==6.29.5 +ipykernel==7.3.0 # via myst-nb -ipython==8.32.0 +ipython==8.39.0 # via # ipykernel # myst-nb -jedi==0.19.2 +jedi==0.20.0 # via ipython -jinja2==3.1.4 +jinja2==3.1.6 # via # myst-parser # sphinx -jsonschema==4.23.0 +jsonschema==4.26.0 # via nbformat -jsonschema-specifications==2024.10.1 +jsonschema-specifications==2025.9.1 # via jsonschema jupyter-cache==1.0.1 # via myst-nb -jupyter-client==8.6.3 +jupyter-client==8.9.1 # via # ipykernel # nbclient -jupyter-core==5.7.2 +jupyter-core==5.9.1 # via # ipykernel # jupyter-client @@ -105,21 +98,21 @@ markdown-it-py==3.0.0 # via # mdit-py-plugins # myst-parser -markupsafe==2.1.5 +markupsafe==3.0.3 # via jinja2 -matplotlib-inline==0.1.7 +matplotlib-inline==0.2.2 # via # ipykernel # ipython -mdit-py-plugins==0.4.1 +mdit-py-plugins==0.6.1 # via myst-parser mdurl==0.1.2 # via markdown-it-py -myst-nb==1.2.0 +myst-nb==1.4.0 # via rocm-docs-core -myst-parser==3.0.1 +myst-parser==4.0.1 # via myst-nb -nbclient==0.10.2 +nbclient==0.11.0 # via # jupyter-cache # myst-nb @@ -128,81 +121,81 @@ nbformat==5.10.4 # jupyter-cache # myst-nb # nbclient -nest-asyncio==1.6.0 +nest-asyncio2==1.7.2 # via ipykernel -packaging==24.1 +packaging==26.2 # via # ipykernel # pydata-sphinx-theme # sphinx -parso==0.8.4 +parso==0.8.7 # via jedi pexpect==4.9.0 # via ipython -platformdirs==4.3.6 +platformdirs==4.10.0 # via jupyter-core -prompt-toolkit==3.0.50 +prompt-toolkit==3.0.52 # via ipython -psutil==7.0.0 +psutil==7.2.2 # via ipykernel ptyprocess==0.7.0 # via pexpect pure-eval==0.2.3 # via stack-data -pycparser==2.22 +pycparser==3.0 # via cffi pydata-sphinx-theme==0.15.4 # via # rocm-docs-core # sphinx-book-theme -pygithub==2.3.0 +pygithub==2.9.1 # via rocm-docs-core -pygments==2.18.0 +pygments==2.20.0 # via # accessible-pygments # ipython # pydata-sphinx-theme # sphinx -pyjwt[crypto]==2.8.0 +pyjwt==2.13.0 # via pygithub -pynacl==1.5.0 +pynacl==1.6.2 # via pygithub python-dateutil==2.9.0.post0 # via jupyter-client -pyyaml==6.0.1 +pyyaml==6.0.3 # via # jupyter-cache # myst-nb # myst-parser # rocm-docs-core # sphinx-external-toc -pyzmq==26.2.1 +pyzmq==27.1.0 # via # ipykernel # jupyter-client -referencing==0.36.2 +referencing==0.37.0 # via # jsonschema # jsonschema-specifications -requests==2.32.3 +requests==2.34.2 # via # pygithub # sphinx -rocm-docs-core==1.18.2 - # via -r requirements.in -rpds-py==0.22.3 +rocm-docs-core @ git+https://github.com/ROCm/rocm-docs-core.git@develop + # via -r ROCmValidationSuite/docs/sphinx/requirements.in +rpds-py==0.30.0 # via # jsonschema # referencing six==1.17.0 # via python-dateutil -smmap==5.0.1 +smmap==5.0.3 # via gitdb -snowballstemmer==2.2.0 +snowballstemmer==3.1.1 # via sphinx -soupsieve==2.5 +soupsieve==2.8.4 # via beautifulsoup4 -sphinx==7.4.5 +sphinx==8.1.3 # via # breathe # myst-nb @@ -213,44 +206,46 @@ sphinx==7.4.5 # sphinx-copybutton # sphinx-design # sphinx-external-toc + # sphinx-multitoc-numbering # sphinx-notfound-page -sphinx-book-theme==1.1.3 +sphinx-book-theme==1.1.4 # via rocm-docs-core sphinx-copybutton==0.5.2 # via rocm-docs-core -sphinx-design==0.6.0 +sphinx-design==0.6.1 # via rocm-docs-core -sphinx-external-toc==1.0.1 +sphinx-external-toc==1.1.0 # via rocm-docs-core -sphinx-notfound-page==1.0.2 +sphinx-multitoc-numbering==0.1.3 + # via sphinx-external-toc +sphinx-notfound-page==1.1.0 # via rocm-docs-core -sphinxcontrib-applehelp==1.0.8 +sphinxcontrib-applehelp==2.0.0 # via sphinx -sphinxcontrib-devhelp==1.0.6 +sphinxcontrib-devhelp==2.0.0 # via sphinx -sphinxcontrib-htmlhelp==2.0.5 +sphinxcontrib-htmlhelp==2.1.0 # via sphinx sphinxcontrib-jsmath==1.0.1 # via sphinx -sphinxcontrib-qthelp==1.0.7 +sphinxcontrib-qthelp==2.0.0 # via sphinx -sphinxcontrib-serializinghtml==1.1.10 +sphinxcontrib-serializinghtml==2.0.0 # via sphinx -sqlalchemy==2.0.38 +sqlalchemy==2.0.50 # via jupyter-cache stack-data==0.6.3 # via ipython -tabulate==0.9.0 +tabulate==0.10.0 # via jupyter-cache -tomli==2.0.1 +tomli==2.4.1 # via sphinx -tornado==6.4.2 +tornado==6.5.7 # via # ipykernel # jupyter-client -traitlets==5.14.3 +traitlets==5.15.1 # via - # comm # ipykernel # ipython # jupyter-client @@ -258,21 +253,24 @@ traitlets==5.14.3 # matplotlib-inline # nbclient # nbformat -typing-extensions==4.12.2 +typing-extensions==4.15.0 # via + # beautifulsoup4 + # cryptography + # exceptiongroup # ipython + # jupyter-client # myst-nb # pydata-sphinx-theme # pygithub + # pyjwt # referencing # sqlalchemy -urllib3==2.2.2 +urllib3==2.7.0 # via # pygithub # requests -wcwidth==0.2.13 +wcwidth==0.8.1 # via prompt-toolkit -wrapt==1.16.0 - # via deprecated -zipp==3.21.0 +zipp==4.1.0 # via importlib-metadata diff --git a/docs/transferbench.md b/docs/transferbench.md new file mode 100644 index 000000000..f8352498a --- /dev/null +++ b/docs/transferbench.md @@ -0,0 +1,129 @@ +# TransferBench in RVS + +[TransferBench](https://github.com/ROCm/TransferBench) is a low-level utility +for measuring host-to-device, device-to-device, and NIC transfer performance on +AMD platforms. It is vendored into RVS as a git submodule at +`external/TransferBench` and built as part of the RVS build. + +## What ships in the RVS package + +Building RVS produces a single DEB or RPM that contains: + +- The `rvs` launcher and all RVS test modules, including `pebb.so` and + `pbqt.so`, which use the TransferBench headers internally as their transfer + backend. +- The standalone `TransferBench` CLI binary, installed alongside `rvs` under + the same prefix (for example `/opt/rocm/extras-/bin/TransferBench`). + +No separate TransferBench package needs to be installed. + +## Why the CLI is bundled + +The `TransferBench` CLI is bundled **for compatibility**. Existing tooling, +scripts, and CI pipelines that invoke `TransferBench` directly continue to +work after installing RVS, without requiring a second package. + +For new work, the recommended entry points are: + +- **RVS** — use the `pebb` and `pbqt` modules with a YAML config to drive + TransferBench through RVS's standard CLI, result schema, and logging. + See the [User guide](ug1main.md) for the supported `transfer_method`, + `transferbench_test`, `executor`, and related options. +- **The TransferBench API** — headers live under + `external/TransferBench/src/header`. Link against the headers from your + own application when you need programmatic access to the transfer engine. + +The standalone CLI should be reserved for reproducing legacy benchmark +invocations or for ad-hoc one-off measurements outside an RVS run. + +## Build options + +The CLI is **off by default** in `build_packages_local.sh` and in CMake (`BUILD_TRANSFERBENCH_CLI=OFF`). CI enables it unless overridden. To build it locally: + +``` +BUILD_TRANSFERBENCH_CLI=ON ./build_packages_local.sh +# or +cmake -DBUILD_TRANSFERBENCH_CLI=ON ... +``` + +Skipping the CLI does not affect RVS's `pebb`/`pbqt` modules — they consume +the TransferBench *headers* from the submodule, not the CLI binary. + +When the CLI is built, it is installed next to `rvs` under the same prefix. The +`TransferBench` binary links `libnuma` and ROCm libraries; package metadata +requires **`libnuma1`** (Debian/Ubuntu and SLES) or **`numactl-libs`** (RHEL/CentOS). +RPM packages declare `(numactl-libs or libnuma1)`. +RUNPATH entries point at the ROCm stack (`/opt/rocm/lib`, `/opt/rocm/core-/lib`). +See [`CMakeTransferBenchRPATH.cmake.in`](../CMakeTransferBenchRPATH.cmake.in). + +### GPU targets vs SDK tarball family + +`GPU_FAMILY` (for example `gfx110X-all`, `gfx1151`) selects which **ROCm SDK +tarball** `build_packages_local.sh` downloads. It does **not** control which GPU +architectures the `TransferBench` binary is compiled for. + +Offload architectures are set by `GPU_TARGETS` / `TRANSFERBENCH_GPU_TARGETS`. +By default RVS uses the same list as upstream TransferBench packaging +(`external/TransferBench/build_packages_local.sh`): + +``` +gfx906;gfx908;gfx90a;gfx942;gfx950;gfx1030;gfx1100;gfx1101;gfx1102;gfx1150;gfx1151;gfx1200;gfx1201 +``` + +Override when building: + +``` +GPU_TARGETS="gfx90a;gfx942" BUILD_TRANSFERBENCH_CLI=ON ./build_packages_local.sh +# or +cmake -DBUILD_TRANSFERBENCH_CLI=ON -DTRANSFERBENCH_GPU_TARGETS="gfx90a;gfx942" ... +``` + +The sub-build also passes `-DHIP_PLATFORM=amd` and disables optional TransferBench +features (NIC executor, MPI, DMA-BUF, local-GPU-only mode) to match upstream +relocatable packaging and keep the bundled CLI dependency-light. + +`GPU_TARGETS` is injected via the ExternalProject initial cache (`-C` file), not +`-DGPU_TARGETS=...` on the command line, because CMake `list(APPEND)` splits +semicolon-separated values into separate list entries. + +CMake logs forwarded args at parent configure time. In **GitHub Actions**, the +TransferBench sub-build runs `cmake --build ... --verbose` with `LOG_BUILD=OFF` so +compile/link lines appear in the workflow log. Local builds keep stamp logs under +`build/TransferBenchCLI-prefix/src/TransferBenchCLI-stamp/`. + +## Submodule layout + +``` +ROCmValidationSuite/ + external/ + TransferBench/ # git submodule, pinned in the RVS tree + src/ + client/Client.cpp # CLI entry point + header/ # public headers consumed by pebb/pbqt +``` + +After cloning RVS, initialise the submodule if you did not pass +`--recurse-submodules`: + +``` +git submodule update --init --recursive +``` + +## Version pinning + +The TransferBench submodule is pinned to a specific commit in the RVS tree, +so a given RVS tag always bundles a known-good TransferBench revision. To +move to a newer TransferBench, update the submodule pointer in a normal pull +request against RVS. + +TransferBench v1.69+ adds `third-party/ibverbs/` headers (`IbvDynLoad.hpp`). +RVS auto-detects that directory at configure time and adds it to the PEBB/PBQT +include path (`TRANSFERBENCH_IBVERBS_INC_DIR` in the root `CMakeLists.txt`). +When the folder is absent (older submodule pins), no extra includes or `libdl` +link are applied. Runtime `libibverbs` is loaded dynamically only if present; +PEBB/PBQT gfx/dma tests do not require it. + +## Further reading + +- [TransferBench project](https://github.com/ROCm/TransferBench) +- [RVS User guide — `pebb` and `pbqt` modules](ug1main.md) diff --git a/docs/ug1main.md b/docs/ug1main.md index 65e57eff5..939bfcb2b 100644 --- a/docs/ug1main.md +++ b/docs/ug1main.md @@ -1,16 +1,15 @@ -# User Guide +# User guide -## Introduction The ROCm Validation Suite (RVS) is a system validation and diagnostics tool for monitoring, stress testing, detecting and troubleshooting issues that -affects the functionality and performance of AMD GPU(s) operating in a +affect the functionality and performance of AMD GPU(s) operating in a high-performance/AI/ML computing environment. RVS is enabled using the ROCm software stack on a compatible software and hardware platform. RVS is a collection of tests, benchmarks, and qualification tools each targeting a specific sub-system of the ROCm platform. The tools are -implemented in software and share a common command line interface. Each set of -tests are implemented in a “module” which is a library encapsulating the +implemented in software and share a common command-line interface. Each set of +tests is implemented in a “module” which is a library encapsulating the functionality specific to the tool. The CLI can specify the directory containing modules to use when searching for libraries to load. Each module may have a set of options that it defines and a configuration file that supports its execution. @@ -20,7 +19,7 @@ of options that it defines and a configuration file that supports its execution. RVS can be obtained by building it from source code base or by installing from pre-built package. -### Building from Source Code +### Building from source code RVS has been developed as open source solution. Its source code and belonging documentation can be found at AMD's GitHub page. @@ -29,50 +28,67 @@ In order to build RVS from source code, refer site](https://github.com/ROCm/ROCmValidationSuite) and follow instructions in README file. -### Installing from Package +### Installing from package Based on the OS, use the appropriate package manager to install the **rocm-validation-suite** package. -For more details, refer [ROCm Validation Suite GitHub site](https://github.com/ROCm/ROCmValidationSuite). +For more details, refer to the [ROCm Validation Suite GitHub site](https://github.com/ROCm/ROCmValidationSuite). RVS package components are installed in `/opt/rocm`. Package contains: -- executable binary (located in _install-base_/bin/rvs) -- public shared libraries (located in _install-base_/lib) -- module specific shared libraries (located in _install-base_/lib/rvs) -- default configuration files (located in _install-base_/share/rocm-validation-suite/conf) -- GPU specific configuration files (located in _install-base_/share/rocm-validation-suite/conf/) -- testscripts (located in _install-base_/share/rocm-validation-suite/testscripts) -- user guide (located in _install-base_/share/rocm-validation-suite/userguide) -- man page (located in _install-base_/share/man) +- executable binary (located in `_install-base_/bin/rvs`) +- public shared libraries (located in `_install-base_/lib`) +- module specific shared libraries (located in `_install-base_/lib/rvs`) +- default configuration files (located in `_install-base_/share/rocm-validation-suite/conf`) +- GPU specific configuration files (located in `_install-base_/share/rocm-validation-suite/conf/`) +- testscripts (located in `_install-base_/share/rocm-validation-suite/testscripts`) +- user guide (located in `_install-base_/share/rocm-validation-suite/userguide`) +- man page (located in `_install-base_/share/man`) ### Running RVS #### Run version built from source code - cd /build/bin +```bash +cd /build/bin +``` Command examples - ./rvs --help ; Lists all options to run RVS test suite - ./rvs -g ; Lists supported GPUs available in the machine - ./rvs -c conf/gst_single.conf ; Run GST module default test configuration -### Run version pre-compiled and packaged with ROCm release +```bash +./rvs --help # List all options +./rvs -g # List supported GPUs available in the machine +./rvs -c conf/gst_single.conf # Run GST module default test configuration +./rvs -m gst # Run GST module using platform-detected config +./rvs -r 3 # Run predefined level 3 tests (range: 1–5, 5 = highest stress) +``` - cd /opt/rocm/bin +#### Run version pre-compiled and packaged with ROCm release + +```bash +cd /opt/rocm/bin +``` Command examples - ./rvs --help ; Lists all options to run RVS test suite - ./rvs -g ; Lists supported GPUs available in the machine - ./rvs -c ../share/rocm-validation-suite/conf/gst_single.conf ; Run GST default test configuration +```bash +./rvs --help # List all options +./rvs -g # List supported GPUs available in the machine +./rvs -c ../share/rocm-validation-suite/conf/gst_single.conf # Run GST default test configuration +./rvs -m gst # Run GST module using platform-detected config +./rvs -r 3 # Run predefined level 3 tests (range: 1–5, 5 = highest stress) +``` -To run GPU specific test configuration, use configuration files from GPU folders in "/opt/rocm/share/rocm-validation-suite/conf" +To run a GPU-specific test configuration, use configuration files from the GPU subfolders under `/opt/rocm/share/rocm-validation-suite/conf/`: - ./rvs -c ../share/rocm-validation-suite/conf/MI300X/gst_single.conf ; Run MI300X specific GST test configuration - ./rvs -c ../share/rocm-validation-suite/conf/nv32/gst_single.conf ; Run Navi 32 specific GST test configuration +```bash +./rvs -c ../share/rocm-validation-suite/conf/MI300X/gst_single.conf # Run MI300X-specific GST test configuration +./rvs -c ../share/rocm-validation-suite/conf/nv32/gst_single.conf # Run Navi 32-specific GST test configuration +``` -Note: If present, always use GPU specific configurations instead of default test configurations. +```{note} +If present, always use GPU specific configurations instead of default test configurations. +``` -## Basic Concepts +## Basic concepts -### RVS Architecture +### RVS architecture RVS is implemented as a set of modules each implementing particular test functionality. Modules are invoked from one central place (aka Launcher) which @@ -82,22 +98,13 @@ architecture is built around concept of Linux shared objects, thus allowing for easy addition of new modules in the future. -### Available Modules +### Available modules -#### GPU Properties – GPUP +#### GPU Properties – GPUP module The GPU Properties module queries the configuration of a target device and returns the device’s static characteristics. These static values can be used to debug issues such as device support, performance and firmware problems. -#### GPU Monitor – GM module -The GPU monitor tool is capable of running on one, some or all of the GPU(s) installed and will report various information at regular intervals. The module can be configured to halt another RVS modules execution if one of the quantities exceeds a specified boundary value. - -#### PCI Express State Monitor – PESM module -The PCIe State Monitor tool is used to actively monitor the PCIe interconnect between the host platform and the GPU. The module will register a “listener” on a target GPU’s PCIe interconnect, and log a message whenever it detects a state change. The PESM will be able to detect the following state changes: - -1. PCIe link speed changes -2. GPU power state changes - #### ROCm Configuration Qualification Tool - RCQT module -The ROCm Configuration Qualification Tool ensures the platform is capable of running ROCm applications and is configured correctly. It checks the installed versions of the ROCm components and the platform configuration of the system. This includes checking the dependencies corresponding to the ROCm meta-packages are installed correctly. +The ROCm Configuration Qualification Tool ensures the platform is capable of running ROCm applications and is configured correctly. It checks the installed versions of the ROCm components and the platform configuration of the system. This includes verifying that the dependencies corresponding to the ROCm meta-packages are installed correctly. #### PCI Express Qualification Tool – PEQT module The PCIe Qualification Tool is used to qualify the PCIe bus on which the GPU is connected. The qualification test will be capable of determining the following characteristics of the PCIe bus interconnect to a GPU: @@ -107,11 +114,6 @@ The PCIe Qualification Tool is used to qualify the PCIe bus on which the GPU is 3. PCIe link speed 4. PCIe link width -#### SBIOS Mapping Qualification Tool – SMQT module -The GPU SBIOS mapping qualification tool is designed to verify that a platform’s SBIOS has satisfied the BAR mapping requirements for VDI and Radeon Instinct products for ROCm support. - -Refer to the “ROCm Use of Advanced PCIe Features and Overview of How BAR Memory is Used In ROCm Enabled System” web page for more information about how BAR memory is initialized by VDI and Radeon products. - #### P2P Benchmark and Qualification Tool – PBQT module The P2P Benchmark and Qualification Tool is designed to provide the list of all GPUs that support P2P and characterize the P2P links between peers. In addition to testing for P2P compatibility, this test will perform a peer-to-peer throughput test between all P2P pairs for performance evaluation. The P2P Benchmark and Qualification Tool will allow users to pick a collection of two or more GPUs on which to run. The user will also be able to select whether or not they want to run the throughput test on each of the pairs. @@ -121,11 +123,18 @@ Please see the web page “ROCm, a New Era in Open GPU Computing” to find out The PCIe Bandwidth Benchmark attempts to saturate the PCIe bus with DMA transfers between system memory and a target GPU card’s memory. The maximum bandwidth obtained is reported to help debug low bandwidth issues. The benchmark should be capable of targeting one, some or all of the GPUs installed in a platform, reporting individual benchmark statistics for each. #### GPU Stress Test - GST module -The GPU Stress Test runs various GEMM computations as workloads to stress the GPU FLOPS performance and check whether it meets the configured target GFLOPS. GEMM workloads shall be configured as either operation type or data type. GEMM based on operation types include SGEMM, DGEMM and HGEMM (Single/Double/Half-precision General Matrix Multiplication) - configured using operation parameter. GEMM based on data types include `fp8`, `i8`, `fp16`, `bf16`, `fp32` and `tf32` (`xf32`) - configured using data type parameter. The duration of the test is configurable, both in terms of time (how long to run) and iterations (how many times to run). +The GPU Stress Test runs various GEMM computations as workloads to stress the GPU FLOPS performance and check whether it meets the configured target GFLOPS. GEMM workloads shall be configured as either operation type or data type. GEMM based on operation types include SGEMM, DGEMM and HGEMM (Single/Double/Half-precision General Matrix Multiplication) - configured using operation parameter. GEMM based on data types include `fp8`, `i8`, `fp16`, `bf16`, `fp32` and `tf32` (`xf32`) - configured using data type parameter. The duration of the test is configurable, both in terms of time (how long to run) and iterations (how many times to run). #### Input EDPp Test - IET module The Input EDPp Test runs GEMM workloads to stress the GPU power (that is, TGP). This test is used to verify if the GPU is capable of handling max. power stress for a sustained period of time. Also checks whether GPU power reaches a set target power. +#### GPU Power Pulse Test - PULSE module +The Pulse test drives repeating **high-power** (GEMM compute) and **low-power** (idle, minimum clocks) phases at a configurable rate so that GPU power swings over time. That pattern stresses the power supply and voltage regulators with transients rather than a single sustained power level. With **parallel: true** on multiple GPUs, the module uses a CPU-side barrier and a GPU-side fine-grained barrier so that devices tend to enter the heavy phase together, increasing aggregate current steps. Power and temperature are read through **AMD SMI**. GEMM execution uses the same **rvs_blas** stack as GST/IET (**rocBLAS** or **hipBLASLt**). + +```{Warning} +This is a beta feature and is not intended for production use. Pass/fail criteria are still being refined. +``` + #### Memory Test - MEM module The Memory module tests the GPU memory for hardware errors and soft errors using HIP. It consists of various tests that use algorithms like Walking 1 bit, Moving inversion and Modulo 20. The module executes the following memory tests [Algorithm, data pattern] @@ -140,105 +149,20 @@ The Memory module tests the GPU memory for hardware errors and soft errors using 9. Modulo 20, random pattern 10. Memory stress test -#### BABEL benchmark Test - BABEL module +#### BABEL Benchmark Test - BABEL module The Babel module executes BabelStream (synthetic GPU benchmark based on the original STREAM benchmark for CPUs) benchmark that measures memory transfer rates (bandwidth) to and from global device memory. Various benchmark tests are implemented using GPU kernels in HIP (Heterogeneous Interface for Portability) programming language. - -### Configuration Files - -The RVS tool will allow the user to indicate a configuration file, adhering to -the YAML 1.2 specification, which details the validation tests to run and the -expected results of a test, benchmark or configuration check. - -The configuration -file used for an execution is specified using the `--config` option. The default -configuration file used for a run is `rvs.conf`, which will include default -values for all defined tests, benchmarks and configurations checks, as well as -device specific configuration values. The format of the configuration files -determines the order in which actions are executed, and can provide the number -of times the test will be executed as well. - -Configuration file is, in YAML terms, mapping of 'actions' keyword into -sequence of action items. Action items are themselves YAML keyed lists. Each -list consists of several _key:value_ pairs. Some keys may have values which -are keyed lists themselves (nested mappings). - -Action item (or action for short) uses keys to define nature of validation test -to be performed. Each action has some common keys -- like 'name', 'module', -'deviceid' -- and test specific keys which depend on the module being used. - -An example of RVS configuration file is given here: - - - actions: - - name: action_1 - device: all - module: gpup - properties: - mem_banks_count: - io_links-properties: - version_major: - - name: action_2 - module: gpup - device: all - properties: - mem_banks_count: - - name: action_3 - ... - - -### Common Configuration Keys - -Common configuration keys applicable to most module are summarized in the -table below:\n - - - - - - - - - - - - - - - -
Config Key Type Description
nameStringThe name of the defined action.
deviceCollection of StringThis is a list of device indexes (gpu ids), or the keyword “all”. The -defined actions will be executed on the specified device, as long as the action -targets a device specifically (some are platform actions). If an invalid device -id value or no value is specified the tool will report that the device was not -found and terminate execution, returning an error regarding the configuration -file.
deviceidIntegerThis is an optional parameter, but if -specified it restricts the action to a specific device type -corresponding to the deviceid.
parallelBoolIf this key is false, actions will be run -on one device at a time, in the order specified in the device list, or the -natural ordering if the device value is “all”. If this parameter is true, -actions will be run on all specified devices in parallel. If a value isn’t -specified the default value is false.
countIntegerThis specifies number of times to execute -the action. If the value is 0, execution will continue indefinitely. If a value -isn’t specified the default is 1. Some modules will ignore this -parameter.
waitIntegerThis indicates how long the test should -wait -between executions, in milliseconds. Some -modules will ignore this parameter. If the -count key is not specified, this key is ignored. -duration Integer This parameter overrides the count key, if -specified. This indicates how long the test -should run, given in milliseconds. Some -modules will ignore this parameter.
moduleStringThis parameter specifies the module that -will be used in the execution of the action. Each module has a set of sub-tests -or sub-actions that can be configured based on its specific -parameters.
- -### Command Line Options +### Command line options Command line options are summarized in the table below: - - +```{note} +Command line options take precedence over the same parameters set in the configuration file. For example, if `parallel: false` is specified in the configuration file but `-p true` is passed on the command line, parallel execution will be enabled. +``` + +
+
Short optionLong option Description
+ @@ -247,6 +171,11 @@ of the current log. Use in conjuction with -d and -l options. This is a mandatory field for test execution. + + @@ -255,9 +184,13 @@ The range is 0-5 with 5 being the highest verbose level. that RVS supports and has visibility. - + + - + + + + -
Short optionLong option Description
-a--appendLogWhen generating a debug logfile, do not overwrite the content of the current log. Use in conjuction with -d and -l options.
-r--runSpecify the predefined test level to run (1–5). +Each level selects a GPU-specific configuration file without requiring a -c argument. +See Test Levels for details. +
-d--debugLevelSpecify the debug level for the output log. The range is 0-5 with 5 being the highest verbose level.
-i--indexesComma separated list of GPU ids/indexes to run test on. -This overrides the device/device_index parameter values specified for every actions in the -configuration file, including the all value. +
-i--indexesComma-separated list of GPU IDs or SMI-based indexes to run tests on. +Overrides the device and device_index values specified in all actions of the configuration file, including the all value. +
-s--selectActionsComma separated list of action names or 0-based action +index numbers to run from the configuration file. Only the matching actions will be executed. +All other actions are skipped.
-j--jsonGenerate output file in JSON format. @@ -268,7 +201,16 @@ else a file created in /var/tmp/ with timestamp in name.
-l--debugLogFileGenerate log file with output and debug information.
-t--listTestsList the test modules present in RVS. +
-m--moduleSpecify a module name to run the corresponding +platform-specific (MI-series GPUs) module configuration file. Valid modules: babel, gpup, +gst, iet, mem, pebb, peqt, pbqt, rcqt. +
-t--durationSpecify the test duration (in seconds) for each action. +Overrides the duration value set in all actions of the configuration file. +
--listTestsList the test modules present in RVS.
-v--verboseEnable detailed logging. Equivalent to specifying -d 5 option. @@ -286,7 +228,7 @@ If no value is provided for the option, it defaults to true. Use this option in conjunction with -c option.
--quietNo console output given. See logs and return +
-q--quietNo console output given. See logs and return code for errors.
--versionDisplays the version information and exits. @@ -296,587 +238,414 @@ code for errors.
+ -## GPUP Module -The GPU properties module provides an interface to easily dump the static -characteristics of a GPU. This information is stored in the sysfs file system -for the kfd, with the following path: +### Test Levels - /sys/class/kfd/kfd/topology/nodes/ +When using the `-r` option, RVS automatically selects a predefined configuration file based on the detected GPU platform (for example, `conf/MI300X/levels/rvs_level_N.conf`). No `-c` argument is needed. The five levels represent progressively increasing test coverage and duration: -Each of the GPU nodes in the directory is identified with a number, -indicating the device index of the GPU. This module will ignore count, duration -or wait key values. - -### Module Specific Keys - - - - - - -
Config Key Type Description
propertiesCollection of StringsThe properties key specifies what configuration property or properties the -query is interested in. Possible values are:\n -all - collect all settings\n -gpu_id\n -cpu_cores_count\n -simd_count\n -mem_banks_count\n -caches_count\n -io_links_count\n -cpu_core_id_base\n -simd_id_base\n -max_waves_per_simd\n -lds_size_in_kb\n -gds_size_in_kb\n -wave_front_size\n -array_count\n -simd_arrays_per_engine\n -cu_per_simd_array\n -simd_per_cu\n -max_slots_scratch_cu\n -vendor_id\n -device_id\n -location_id\n -drm_render_minor\n -max_engine_clk_fcompute\n -local_mem_size\n -fw_version\n -capability\n -max_engine_clk_ccompute\n -
io_links-propertiesCollection of StringsThe properties key specifies what configuration -property or properties the query is interested in. -Possible values are:\n -all - collect all settings\n -count - the number of io_links\n -type\n -version_major\n -version_minor\n -node_from\n -node_to\n -weight\n -min_latency\n -max_latency\n -min_bandwidth\n -max_bandwidth\n -recommended_transfer_size\n -flags\n -
- -### Output - -Module specific output keys are described in the table below: - - - - - - +
+
Output Key Type Description
properties-valuesCollection of IntegersThe collection will contain a positive integer value for each of the valid -properties specified in the properties config key.
io_links-propertiesvaluesCollection of IntegersThe collection will contain a positive integer value for each of the valid -properties specified in the io_links-properties config key.
+ + + + + +
LevelTestDescriptionModulesTypical Duration
1Software validationVerifies that all required ROCm packages and metapackages are correctly installed.rcqtShort (seconds)
2Hardware capability checksChecks PCIe link speed and width, performs a quick HBM read/write pass, and verifies basic PCIe and XGMI interface connectivity.peqt, babel, pebb, pbqtShort (seconds)
3Basic performanceMeasures HBM bandwidth across common operations, PCIe host-to-device and device-to-host throughput, bidirectional XGMI bandwidth, and limited gemm compute performance runs.babel, pebb, pbqt, gstMinutes
4Extended performanceSame coverage as level 3 with longer test durations, full HBM operation set, bidirectional PCIe transfers, a basic memory test pass, and full gemm compute performance runs.babel, pebb, pbqt, mem, gst, ietTens of minutes
5Full stress testRuns all tests from level 4 at maximum duration, adds high-iteration memory stress testing, and includes a sustained power stress test targeting peak GPU TDP. Intended for thorough pre-production validation.babel, pebb, pbqt, mem, gst, ietHours
-Each of the settings specified has a positive integer value. For each -setting requested in the properties key a message with the following format will -be returned: + - [RESULT][][] gpup +The exact tests and thresholds within each level vary by GPU platform. -For each setting in the io_links-properties key a message with the following -format will be returned: +#### Examples - [RESULT][][] gpup +Print version information and exit: +```bash +./rvs --version +``` -### Examples +List all available test modules: +```bash +./rvs --listTests +``` -**Example 1:** +Override the duration for all actions to 30 seconds: +```bash +./rvs -c conf/gst_single.conf -t 30 +``` +List all GPUs visible to RVS: +```bash +./rvs -g +``` -Consider action> +Run a specific configuration file: +```bash +./rvs -c conf/gst_single.conf +``` - actions: - - name: action_1 - device: all - module: gpup - properties: - all: - io_links-properties: - all: +Run a platform-specific configuration file (for example, MI350X GST): +```bash +./rvs -c conf/MI350X/gst_single.conf +``` -Action will display all properties for all compatible GPUs present in the -system. Output for such configuration may be like this: - - [RESULT] [597737.498442] action_1 gpup 3254 cpu_cores_count 0 - [RESULT] [597737.498517] action_1 gpup 3254 simd_count 256 - [RESULT] [597737.498558] action_1 gpup 3254 mem_banks_count 1 - [RESULT] [597737.498598] action_1 gpup 3254 caches_count 96 - [RESULT] [597737.498637] action_1 gpup 3254 io_links_count 1 - [RESULT] [597737.498680] action_1 gpup 3254 cpu_core_id_base 0 - [RESULT] [597737.498725] action_1 gpup 3254 simd_id_base 2147487744 - [RESULT] [597737.498768] action_1 gpup 3254 max_waves_per_simd 10 - [RESULT] [597737.498812] action_1 gpup 3254 lds_size_in_kb 64 - [RESULT] [597737.498856] action_1 gpup 3254 gds_size_in_kb 0 - [RESULT] [597737.498901] action_1 gpup 3254 wave_front_size 64 - [RESULT] [597737.498945] action_1 gpup 3254 array_count 4 - [RESULT] [597737.498990] action_1 gpup 3254 simd_arrays_per_engine 1 - [RESULT] [597737.499035] action_1 gpup 3254 cu_per_simd_array 16 - [RESULT] [597737.499081] action_1 gpup 3254 simd_per_cu 4 - [RESULT] [597737.499128] action_1 gpup 3254 max_slots_scratch_cu 32 - [RESULT] [597737.499175] action_1 gpup 3254 vendor_id 4098 - [RESULT] [597737.499222] action_1 gpup 3254 device_id 26720 - [RESULT] [597737.499270] action_1 gpup 3254 location_id 8960 - [RESULT] [597737.499318] action_1 gpup 3254 drm_render_minor 128 - [RESULT] [597737.499369] action_1 gpup 3254 max_engine_clk_ccompute 2200 - [RESULT] [597737.499419] action_1 gpup 3254 local_mem_size 17163091968 - [RESULT] [597737.499468] action_1 gpup 3254 fw_version 405 - [RESULT] [597737.499518] action_1 gpup 3254 capability 8832 - [RESULT] [597737.499569] action_1 gpup 3254 max_engine_clk_ccompute 2200 - [RESULT] [597737.499633] action_1 gpup 3254 0 count 1 - [RESULT] [597737.499675] action_1 gpup 3254 0 type 2 - [RESULT] [597737.499695] action_1 gpup 3254 0 version_major 0 - [RESULT] [597737.499716] action_1 gpup 3254 0 version_minor 0 - [RESULT] [597737.499736] action_1 gpup 3254 0 node_from 4 - [RESULT] [597737.499763] action_1 gpup 3254 0 node_to 1 - [RESULT] [597737.499783] action_1 gpup 3254 0 weight 20 - [RESULT] [597737.499808] action_1 gpup 3254 0 min_latency 0 - [RESULT] [597737.499830] action_1 gpup 3254 0 max_latency 0 - [RESULT] [597737.499853] action_1 gpup 3254 0 min_bandwidth 0 - [RESULT] [597737.499878] action_1 gpup 3254 0 max_bandwidth 0 - [RESULT] [597737.499902] action_1 gpup 3254 0 recommended_transfer_size 0 - [RESULT] [597737.499927] action_1 gpup 3254 0 flags 1 - [RESULT] [597737.500208] action_1 gpup 50599 cpu_cores_count 0 - [RESULT] [597737.500254] action_1 gpup 50599 simd_count 256 - ... - [RESULT] [597737.501603] action_1 gpup 50599 0 recommended_transfer_size 0 - [RESULT] [597737.501626] action_1 gpup 50599 0 flags 1 - [RESULT] [597737.501877] action_1 gpup 33367 cpu_cores_count 0 - [RESULT] [597737.501921] action_1 gpup 33367 simd_count 256 - ... - [RESULT] [597737.503258] action_1 gpup 33367 0 recommended_transfer_size 0 - [RESULT] [597737.503282] action_1 gpup 33367 0 flags 1 - ... +Repeat a test 5 times in sequence (soak/stability testing): +```bash +./rvs -c conf/gst_single.conf -n 5 +``` -**Example 2:** +Run only the GST module using the built-in platform configuration: +```bash +./rvs -m gst +``` -Consider action: - actions: - - name: action_1 - device: all - module: gpup - properties: - simd_count: - mem_banks_count: - io_links_count: - vendor_id: - device_id: - location_id: - max_engine_clk_ccompute: - io_links-properties: - version_major: - type: - version_major: - version_minor: - node_from: - node_to: - recommended_transfer_size: - flags: - -This action explicitly lists some of the properties. -Output for such configuration may be: - - [RESULT] [597868.690637] action_1 gpup 3254 device_id 26720 - [RESULT] [597868.690713] action_1 gpup 3254 io_links_count 1 - [RESULT] [597868.690766] action_1 gpup 3254 location_id 8960 - [RESULT] [597868.690819] action_1 gpup 3254 max_engine_clk_ccompute 2200 - [RESULT] [597868.690862] action_1 gpup 3254 mem_banks_count 1 - [RESULT] [597868.690903] action_1 gpup 3254 simd_count 256 - [RESULT] [597868.690950] action_1 gpup 3254 vendor_id 4098 - [RESULT] [597868.691029] action_1 gpup 3254 0 flags 1 - [RESULT] [597868.691053] action_1 gpup 3254 0 node_from 4 - [RESULT] [597868.691075] action_1 gpup 3254 0 node_to 1 - [RESULT] [597868.691099] action_1 gpup 3254 0 recommended_transfer_size 0 - [RESULT] [597868.691119] action_1 gpup 3254 0 type 2 - [RESULT] [597868.691138] action_1 gpup 3254 0 version_major 0 - [RESULT] [597868.691158] action_1 gpup 3254 0 version_minor 0 - [RESULT] [597868.691425] action_1 gpup 50599 device_id 26720 - [RESULT] [597868.691469] action_1 gpup 50599 io_links_count 1 - [RESULT] [597868.691517] action_1 gpup 50599 location_id 17152 - ... - [RESULT] [597868.692159] action_1 gpup 33367 device_id 26720 - [RESULT] [597868.692204] action_1 gpup 33367 io_links_count 1 - [RESULT] [597868.692252] action_1 gpup 33367 location_id 25344 - ... - [RESULT] [597868.692619] action_1 gpup 33367 0 version_minor 0 +Run the predefined level 3 test suite (basic performance): +```bash +./rvs -r 3 +``` -**Example 3:** +Run the predefined level 5 (full stress) test suite on specific GPUs only: +```bash +./rvs -r 5 -i 0,1 +``` -Consider this action: +Run a configuration file on specific GPUs (by index) with parallel execution enabled: +```bash +./rvs -c conf/gst_single.conf -i 0,1 -p true +``` - actions: - - name: action_1 - device: all - module: gpup - deviceid: 267 - properties: - all: - io_links-properties: - all: +Run only selected actions from a configuration file (by name or 0-based index): +```bash +./rvs -c conf/iet_single.conf -s action_1,action_2 +./rvs -c conf/iet_single.conf -s 0,2 +``` -Action lists deviceid 267 which is not present in the system. -Output for such configuration is: - - RVS-GPUP: action: action_1 invalid 'deviceid' key value - - - - -## GM Module -The GPU monitor module can be used monitor and characterize the response of a -GPU to different levels of use. This module is intended to run concurrently with -other actions, and provides a ‘start’ and ‘stop’ configuration key to start the -monitoring and then stop it after testing has completed. The module can also be -configured with bounding box values for interested GPU parameters. If any of the -GPU’s parameters exceed the bounding values on a specific GPU an INFO warning -message will be printed to stdout while the bounding value is still exceeded. - -### Module Specific Keys - - - - - - - - - - - - - - -
Config Key Type Description
monitorBoolIf this this key is set to true, the GM module will start monitoring on -specified devices. If this key is set to false, all other keys are ignored and -monitoring of the specified device will be stopped.
metricsCollection of Structures, specifying the metric, if there are bounds and the -bound values. The structures have the following format:\n{String, Bool, Integer, -Integer}The set of metrics to monitor during the monitoring period. Example values -are:\n{‘temp’, ‘true’, max_temp, min_temp}\n {‘clock’, ‘false’, max_clock, -min_clock}\n {‘mem_clock’, ‘true’, max_mem_clock, min_mem_clock}\n {‘fan’, -‘true’, max_fan, min_fan}\n {‘power’, ‘true’, max_power, min_power}\n The set of -upper bounds for each metric are specified as an integer. The units and values -for each metric are:\n temp - degrees Celsius\n clock - MHz \n mem_clock - MHz -\n fan - Integer between 0 and 255 \n power - Power in Watts
sample_intervalIntegerIf this key is specified metrics will be sampled at the given rate. The -units for the sample_interval are milliseconds. The default value is 1000. -
log_intervalIntegerIf this key is specified informational messages will be emitted at the given -interval, providing the current values of all parameters specified. This -parameter must be equal to or greater than the sample rate. If this value is not -specified, no logging will occur.
terminateBool If the terminate key is true the GM -monitor will terminate the RVS process when a bounds violation is encountered on -any of the metrics specified.
forceBool If 'true' and terminate key is also 'true' -the RVS process will terminate immediately. **Note:** this may cose resource leaks -within GPUs.
+Generate JSON output to a specific file: +```bash +./rvs -c conf/gst_single.conf -j /tmp/rvs_results.json +``` -### Output +Run with verbose logging and save the log to a file: +```bash +./rvs -c conf/gst_single.conf -v -l /tmp/rvs_debug.log +``` -Module specific output keys are described in the table below: - - - - - -
Output Key Type Description
metric_valuesTime Series Collection of Result -IntegersA collection of integers containing the result values for each -of the metrics being monitored.
metric_violationsCollection of Result Integers A -collection of integers containing the violation count for each of the metrics -being monitored.
metric_averageCollection of Result Integers
+Suppress console output and write results to a JSON file (useful in CI pipelines): +```bash +./rvs -c conf/gst_single.conf --quiet -j /tmp/rvs_results.json +``` -When monitoring is started for a target GPU, a result message is logged -with the following format: +The configuration files in the top-level `conf/` folder are generic samples and may not reflect the correct thresholds or parameters for your hardware. For accurate results, use the platform-specific configuration files under `conf//` (for example, `conf/MI350X/`), or use the `-r` or `-m` option, which automatically selects the appropriate platform configuration. - [RESULT][][] gm started +### GPU-Specific Configurations -In addition, an informational message is provided for each for each metric -being monitored: +RVS includes optimized test configurations for a range of GPU families, organized under `conf//`. +Level-based configurations (usable with `-r`) are available for: MI300X, MI300X-HF, MI308X, MI308X-HF, MI325X, MI350X, MI355X, MI350P-450W, MI350P-600W, MI450X, nv21, nv31, nv32, gfx1200, gfx1201, RX9060, RX9070, RX9070GRE, R9600D. - [INFO ][][] gm monitoring bounds min: max: +Radeon level suites omit HBM (babel) and XGMI (pbqt) tests that do not apply to consumer GPUs. GST `target_stress` and IET `target_power` values are sourced from each platform's `gst_single.conf` and `iet_single.conf`. -During the monitoring informational output regarding the metrics of the GPU will -be sampled at every interval specified by the sample_rate key. If a bounding box -violation is discovered during a sampling interval, a warning message is -logged with the following format: +Navi 48 SKUs (RX9070, RX9070GRE, R9600D, gfx1201) share PCI device IDs `0x7550`/`0x7551`; RVS defaults to `RX9070`. Use `-c conf//levels/rvs_level_N.conf` explicitly for other Navi 48 SKUs if auto-detection selects the wrong folder. - [INFO ][][] gm bounds violation +For MI350P, the 450W and 600W SKUs share a single PCI device ID (`0x75a8`); RVS disambiguates between them at runtime by reading the GPU's hardware power cap via amdsmi (~450 W → `MI350P-450W`, ~600 W → `MI350P-600W`). If the power cap cannot be read, RVS falls back to `MI350P-450W` and prints a warning; use `-c conf/MI350P-600W/levels/rvs_level_N.conf` explicitly if your card is the 600W SKU. -If the log_interval value is set an information message for each metric is -logged at every interval using the following format: +**AMD Instinct Series:** +- MI210, MI250X, MI300A, MI300X, MI308X, MI325X, MI350X, MI355X +- High-frequency variants: MI300X-HF, MI308X-HF - [INFO ][][] gm +**AMD Radeon Series:** +- nv21, nv31, nv32, gfx1200, gfx1201 +- RX9060, RX9070, RX9070GRE, R9600D -When monitoring is stopped for a target GPU, a result message is logged -with the following format: +### Configuration files - [RESULT][][] gm gm stopped +The RVS tool will allow the user to indicate a configuration file, adhering to +the YAML 1.2 specification, which details the validation tests to run and the +expected results of a test, benchmark or configuration check. -The following messages, reporting the number of metric violations that were -sampled over the duration of the monitoring and the average metric value is -reported: +The configuration +file used for an execution is specified using the `--config` option. The default +configuration file used for a run is `rvs.conf`, which will include default +values for all defined tests, benchmarks and configurations checks, as well as +device specific configuration values. The format of the configuration files +determines the order in which actions are executed, and can provide the number +of times the test will be executed as well. - [RESULT][][] gm violations - [RESULT][][] gm average +Configuration file is, in YAML terms, mapping of 'actions' keyword into +sequence of action items. Action items are themselves YAML keyed lists. Each +list consists of several _key:value_ pairs. Some keys may have values which +are keyed lists themselves (nested mappings). -### Examples +Action item (or action for short) uses keys to define nature of validation test +to be performed. Each action has some common keys -- like 'name', 'module', +'deviceid' -- and test specific keys which depend on the module being used. -**Example 1:** +An example of RVS configuration file is given here: -Consider action: actions: - name: action_1 - module: gm device: all - monitor: true - metrics: - temp: true 20 0 - fan: true 10 0 - duration: 5000 - - name: another_action - ... - -This action will monitor temperature and fan speed for 5 seconds and then continue -with the next action. Output for such configuration may be: - - [RESULT] [694381.521373] [action_1] gm 33367 started - [INFO ] [694381.531803] action_1 gm 33367 monitoring temp bounds min:0 max:20 - [INFO ] [694381.531817] action_1 gm 33367 monitoring temp bounds min:0 max:20 - [INFO ] [694381.531828] action_1 gm 33367 monitoring fan bounds min:0 max:10 - [RESULT] [694381.521373] [action_1] gm 3254 started - [INFO ] [694381.532257] action_1 gm 3254 monitoring temp bounds min:0 max:20 - [INFO ] [694381.532276] action_1 gm 3254 monitoring temp bounds min:0 max:20 - [INFO ] [694381.532293] action_1 gm 3254 monitoring fan bounds min:0 max:10 - [RESULT] [694381.521373] [action_1] gm 50599 started - [INFO ] [694381.534471] action_1 gm 50599 monitoring temp bounds min:0 max:20 - [INFO ] [694381.534487] action_1 gm 50599 monitoring temp bounds min:0 max:20 - [INFO ] [694381.534502] action_1 gm 50599 monitoring fan bounds min:0 max:10 - [INFO ] [694381.534623] action_1 gm 33367 temp bounds violation 22C - [INFO ] [694381.534822] action_1 gm 3254 temp bounds violation 22C - [INFO ] [694381.534946] action_1 gm 50599 temp bounds violation 22C - [INFO ] [694382.535329] action_1 gm 33367 temp bounds violation 22C - ... - [INFO ] [694385.537777] action_1 gm 50599 temp bounds violation 21C - [RESULT] [694386.538037] [action_1] gm 3254 stopped - [RESULT] [694386.538037] [action_1] gm 50599 stopped - [RESULT] [694386.538037] [action_1] gm 33367 stopped - [RESULT] [694386.521449] [action_1] gm 3254 temp violations 1 - [RESULT] [694386.521449] [action_1] gm 3254 temp average 19C - [RESULT] [694386.521449] [action_1] gm 3254 fan violations 0 - [RESULT] [694386.521449] [action_1] gm 3254 fan average 0% - [RESULT] [694386.521449] [action_1] gm 50599 temp violations 5 - [RESULT] [694386.521449] [action_1] gm 50599 temp average 21C - [RESULT] [694386.521449] [action_1] gm 50599 fan violations 0 - [RESULT] [694386.521449] [action_1] gm 50599 fan average 0% - [RESULT] [694386.521449] [action_1] gm 33367 temp violations 5 - [RESULT] [694386.521449] [action_1] gm 33367 temp average 22C - [RESULT] [694386.521449] [action_1] gm 33367 fan violations 0 - [RESULT] [694386.521449] [action_1] gm 33367 fan average 0% - -**Example 2:** - -Consider action: - - actions: - - name: action_1 - module: gm + module: gpup + properties: + mem_banks_count: + io_links-properties: + version_major: + - name: action_2 + module: gpup device: all - monitor: true - metrics: - temp: true 20 0 - fan: true 10 0 - power: true 100 0 - sample_interval: 1000 - log_interval: 1200 - terminate: false - duration: 5000 - -This configuration is similar to that in *Example 1* but has explicitly -given values for *sample_interval* and *log_interval*. Output is similar to -the previous one but averaging and the printout are performed at a different -rate. + properties: + mem_banks_count: + - name: action_3 + ... -**Example 3:** -Consider action with syntax error ('temp' key is missing lower value): +### Common configuration keys - actions: - - name: action_1 - module: gm - device: 33367 50599 - monitor: true - metrics: - temp: true 20 - fan: true 10 0 - power: true 100 0 - sample_interval: 1000 - log_interval: 1200 +Common configuration keys applicable to most module are summarized in the +table below: -Output for such configuration is: +
+ + + + + - RVS-GM: action: action_1 Wrong number of metric parameters + -**Example 4:** + -Consider action with logical error: + - actions: - - name: action_1 - module: gm - device: all - monitor: true - metrics: - temp: false 20 0 - clock: true 1500 852 - power: true 100 0 - sample_interval: 5000 - log_interval: 4000 - duration: 8000 + -Output for such configuration is: + - RVS-GM: action: action_1 Log interval has the lower value than the sample interval + -## PESM Module -The PCIe State Monitor (PESM) tool is used to actively monitor the PCIe -interconnect between the host platform and the GPU. The module registers -“listener” on a target GPUs PCIe interconnect, and log a message whenever it -detects a state change. The PESM is able to detect the following state changes: + -1. PCIe link speed changes -2. GPU device power state changes + +
Config Key Type Description
nameStringThe name of the defined action.
deviceCollection of StringThis is a list of device indexes (gpu ids), or the keyword “all”. The +defined actions will be executed on the specified device, as long as the action +targets a device specifically (some are platform actions). If an invalid device +id value or no value is specified the tool will report that the device was not +found and terminate execution, returning an error regarding the configuration +file.
deviceidIntegerThis is an optional parameter, but if +specified it restricts the action to a specific device type +corresponding to the deviceid.
device_indexIntegerThis is an optional parameter that +restricts the action to a GPU identified by its SMI index. Can also be +overridden via the -i CLI option.
parallelBoolIf this key is false, actions will be run +on one device at a time, in the order specified in the device list, or the +natural ordering if the device value is “all”. If this parameter is true, +actions will be run on all specified devices in parallel. If a value isn’t +specified the default value is false.
countIntegerThis specifies number of times to execute +the action. If the value is 0, execution will continue indefinitely. If a value +isn’t specified the default is 1. Some modules will ignore this +parameter.
waitIntegerThis indicates how long the test should +wait between executions, in milliseconds. The default value is 500 ms. +Some modules will ignore this parameter.
durationIntegerThis indicates how long the test +should run, given in milliseconds. When specified, it takes precedence +over the count key for modules that support it. Some modules will +ignore this parameter.
moduleStringThis parameter specifies the module that +will be used in the execution of the action. Each module has a set of sub-tests +or sub-actions that can be configured based on its specific +parameters.
log_intervalIntegerThis specifies how often, in +milliseconds, the module emits a progress or status log message during a +running test. If a value isn't specified the default is 1000 ms. Some modules +will ignore this parameter.
+
-This module is intended to run concurrently with other actions, and provides a -‘start’ and ‘stop’ configuration key to start the monitoring and then stop it -after testing has completed. For information on GPU power state monitoring -please consult the 7.6. PCI Power Management Capability Structure, Gen 3 spec, -page 601, device states D0-D3. For information on link status changes please -consult the 7.8.8. Link Status Register (Offset 12h), Gen 3 spec, page 635. +## GPUP module +The GPU properties module provides an interface to easily dump the static +characteristics of a GPU. This information is stored in the sysfs file system +for the kfd, with the following path: -Monitoring is performed by polling respective PCIe registers roughly every 1ms -(one millisecond). + /sys/class/kfd/kfd/topology/nodes/ -### Module Specific Keys - - - -
Config Key Type Description
monitorBoolThis this key is set to true, the PESM -module will start monitoring on specified devices. If this key is set to false, -all other keys are ignored and monitoring will be stopped for all devices.
+Each of the GPU nodes in the directory is identified with a number, +indicating the device index of the GPU. Use the `device_index` common key to +target specific GPUs by their topology node index. + + +### Module specific keys + +
+ + + + + + + + + + + + + + + + + + + + +
Config KeyTypeDescription
propertiesCollection of Strings + The properties key specifies what configuration property or properties the + query is interested in. Possible values are: +
    +
  • all - collect all settings
  • +
  • gpu_id
  • +
  • cpu_cores_count
  • +
  • simd_count
  • +
  • mem_banks_count
  • +
  • caches_count
  • +
  • io_links_count
  • +
  • cpu_core_id_base
  • +
  • simd_id_base
  • +
  • max_waves_per_simd
  • +
  • lds_size_in_kb
  • +
  • gds_size_in_kb
  • +
  • wave_front_size
  • +
  • array_count
  • +
  • simd_arrays_per_engine
  • +
  • cu_per_simd_array
  • +
  • simd_per_cu
  • +
  • max_slots_scratch_cu
  • +
  • vendor_id
  • +
  • device_id
  • +
  • location_id
  • +
  • drm_render_minor
  • +
  • max_engine_clk_fcompute
  • +
  • local_mem_size
  • +
  • fw_version
  • +
  • capability
  • +
  • max_engine_clk_ccompute
  • +
+
io_links-propertiesCollection of Strings + The properties key specifies what configuration property or properties the + query is interested in. Possible values are: +
    +
  • all - collect all settings
  • +
  • count - the number of io_links
  • +
  • type
  • +
  • version_major
  • +
  • version_minor
  • +
  • node_from
  • +
  • node_to
  • +
  • weight
  • +
  • min_latency
  • +
  • max_latency
  • +
  • min_bandwidth
  • +
  • max_bandwidth
  • +
  • recommended_transfer_size
  • +
  • flags
  • +
+
+
### Output Module specific output keys are described in the table below: - - - -
Output Key Type Description
stateStringA string detailing the current power state -of the GPU or the speed of the PCIe link.
- -When monitoring is started for a target GPU, a result message is logged -with the following format: - [RESULT][][] pesm started +
+ + + + + + +
Output Key Type Description
properties-valuesCollection of IntegersThe collection will contain a positive integer value for each of the valid +properties specified in the properties config key.
io_links-propertiesvaluesCollection of IntegersThe collection will contain a positive integer value for each of the valid +properties specified in the io_links-properties config key.
+
-When monitoring is stopped for a target GPU, a result message is logged -with the following format: +Each of the settings specified has a positive integer value. For each +setting requested in the properties key a message with the following format will +be returned: - [RESULT][][] pesm all stopped + [RESULT][][] gpup -When monitoring is enabled, any detected state changes in link speed or GPU -power state will generate the following informational messages: +For each setting in the io_links-properties key a message with the following +format will be returned: - [INFO ][][] pesm power state change - [INFO ][][] pesm link speed change + [RESULT][][] gpup ### Examples -**Example 1** +**Example:** -Here is a typical check utilizing PESM functionality: +Run: - actions: - - name: action_1 - device: all - module: pesm - monitor: true - - name: action_2 - device: 33367 - module: gst - parallel: false - count: 2 - wait: 100 - duration: 18000 - ramp_interval: 7000 - log_interval: 1000 - max_violations: 1 - copy_matrix: false - target_stress: 5000 - tolerance: 0.07 - matrix_size: 5760 - - name: action_3 - device: all - module: pesm - monitor: false - -- **action_1** will initiate monitoring on all devices by setting key **monitor** to **true**\n -- **action_2** will start GPU stress test -- **action_3** will stop monitoring - -If executed like this: - - sudo rvs -c conf/pesm8.conf -d 3 - -output similar to this one can be produced: - - [RESULT] [497544.637462] [action_1] pesm all started - [INFO ] [497544.648299] [action_1] pesm 33367 link speed change 8 GT/s - [INFO ] [497544.648299] [action_1] pesm 33367 power state change D0 - [INFO ] [497544.648733] [action_1] pesm 3254 link speed change 8 GT/s - [INFO ] [497544.648733] [action_1] pesm 3254 power state change D0 - [INFO ] [497544.650413] [action_1] pesm 50599 link speed change 8 GT/s - [INFO ] [497544.650413] [action_1] pesm 50599 power state change D0 - [INFO ] [497545.170392] [action_2] gst 33367 start 5000.000000 copy matrix:false - [INFO ] [497547.36602 ] [action_2] gst 33367 Gflops 6478.066983 - [INFO ] [497548.69221 ] [action_2] gst 33367 target achieved 5000.000000 - [INFO ] [497549.101219] [action_2] gst 33367 Gflops 5189.993529 - [INFO ] [497550.132376] [action_2] gst 33367 Gflops 5189.993529 - ... - [INFO ] [497563.569370] [action_2] gst 33367 Gflops 5174.935520 - [RESULT] [497564.86904 ] [action_2] gst 33367 Gflop: 6478.066983 flops_per_op: 382.205952x1e9 bytes_copied_per_op: 398131200 try_ops_per_sec: 13.081952 pass: TRUE - [INFO ] [497564.220311] [action_2] gst 33367 start 5000.000000 copy matrix:false - [INFO ] [497566.70585 ] [action_2] gst 33367 Gflops 6521.049418 - [INFO ] [497567.99929 ] [action_2] gst 33367 target achieved 5000.000000 - [INFO ] [497568.143096] [action_2] gst 33367 Gflops 5130.281235 - ... - [INFO ] [497582.683893] [action_2] gst 33367 Gflops 5135.204729 - [RESULT] [497583.130945] [action_2] gst 33367 Gflop: 6521.049418 flops_per_op: 382.205952x1e9 bytes_copied_per_op: 398131200 try_ops_per_sec: 13.081952 pass: TRUE - [RESULT] [497583.155470] [action_3] pesm all stopped + ./rvs -c conf/gpup_single.conf -**Example 2:** - -Consider this file: +Configuration (`conf/gpup_single.conf`, first action): actions: - - name: act1 + - name: RVS-GPUP-TC1 device: all - deviceid: xxx - module: pesm - monitor: true - - -This file has and invalid entry in **deviceid** key. -If execute, an error will be reported: - - RVS-PESM: action: act1 invalide 'deviceid' key value: xxx - + module: gpup + properties: + all: + io_links-properties: + all: -## RCQT Module +Sample output (first action, abridged): + + [RESULT] [302733.913494] Action name :RVS-GPUP-TC1 + [RESULT] [302733.975662] Module name :gpup + [RESULT] [302733.975835] [RVS-GPUP-TC1] gpup 42583 cpu_cores_count 0 + [RESULT] [302733.975836] [RVS-GPUP-TC1] gpup 42583 simd_count 1024 + [RESULT] [302733.975836] [RVS-GPUP-TC1] gpup 42583 mem_banks_count 1 + [RESULT] [302733.975836] [RVS-GPUP-TC1] gpup 42583 caches_count 550 + [RESULT] [302733.975837] [RVS-GPUP-TC1] gpup 42583 io_links_count 8 + [RESULT] [302733.975837] [RVS-GPUP-TC1] gpup 42583 p2p_links_count 1 + [RESULT] [302733.975837] [RVS-GPUP-TC1] gpup 42583 cpu_core_id_base 0 + [RESULT] [302733.975838] [RVS-GPUP-TC1] gpup 42583 simd_id_base 2147487744 + ... + [RESULT] [302733.976401] [RVS-GPUP-TC1] gpup 57875 7 version_major 0 + [RESULT] [302733.976401] [RVS-GPUP-TC1] gpup 57875 7 version_minor 0 + [RESULT] [302733.976402] [RVS-GPUP-TC1] gpup 57875 7 node_from 9 + [RESULT] [302733.976402] [RVS-GPUP-TC1] gpup 57875 7 node_to 8 + [RESULT] [302733.976402] [RVS-GPUP-TC1] gpup 57875 7 weight 15 + [RESULT] [302733.976403] [RVS-GPUP-TC1] gpup 57875 7 min_latency 0 + [RESULT] [302733.976403] [RVS-GPUP-TC1] gpup 57875 7 max_latency 0 + [RESULT] [302733.976403] [RVS-GPUP-TC1] gpup 57875 7 min_bandwidth 0 + [RESULT] [302733.976403] [RVS-GPUP-TC1] gpup 57875 7 recommended_transfer_size 0 + [RESULT] [302733.976404] [RVS-GPUP-TC1] gpup 57875 7 flags 1 + + +=====================================================================+ + | ROCm Validation Suite (RVS) Summary | + +=====================================================================+ + | System Overview | + +---------------------------------------------------------------------+ + | Operating System | Ubuntu 22.04.5 LTS | + | RVS version | 1.6.75 | + | ROCm version | 7.2.1-81 | + | amdgpu version | 6.16.13 | + | GPUs | 8 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 42583 | AMD Instinct MI355X - 27226 | + | 0 - 2 - 0000:05:00.0 | 1 - 3 - 0000:15:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 36479 | AMD Instinct MI355X - 17010 | + | 2 - 4 - 0000:65:00.0 | 3 - 5 - 0000:75:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 1590 | AMD Instinct MI355X - 51771 | + | 4 - 6 - 0000:85:00.0 | 5 - 7 - 0000:95:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 11806 | AMD Instinct MI355X - 57875 | + | 6 - 8 - 0000:e5:00.0 | 7 - 9 - 0000:f5:00.0 | + +=====================================================================+ + | Action Name | Module | Result | + +=====================================================================+ + | RVS-GPUP-TC1 | GPUP | PASS | + +---------------------------------------------------------------------+ + + +## RCQT module RCQT ensures the platform is capable of running ROCm applications and is @@ -888,16 +657,16 @@ The purpose of the RCQT is to provide an extensible, OS independent and scriptable interface capable for performing the configuration checks required for ROCm support. The checks in this module do not target a specific device. -\n\n -Two types of actions are performed by RCQT. -1)Metapackage Check -metapackage-validation: This will check the installation of the mentioned -metapackages and their dependencies and their respective versions as required -by metapackage. List of metapackages are provided with key **package** +Two types of actions are performed by RCQT, selected by which configuration +key is present — not by the action name: + +1) **Metapackage Check** — triggered when the `package` key is present. +Checks the installation of the named metapackages and their dependencies, +verifying the installed versions. + +2) **Packages installation check** — triggered when `rpmpackagelist` or +`debpackagelist` is present. Checks whether the listed packages are installed. -2)Packages installation check -packagelist-install-validation: This action checks if the package is installed. - Packages are provided against key **rpmpackagelist** and **debpackagelist** This feature is used to check installed packages on the system. It provides checks for installed packages and the currently available package versions, if @@ -907,19 +676,22 @@ applicable. Input keys are described in the table below: - - +
+
Config Key Type Description
+
Config Key Type Description
packageCollection of Strings Specifies the list of metapackages to check. This key is required.
+ #### Output Output keys are described in the table below for each metapackage along with versions of each sub package: - - +
+
Output Key Type Description
+ @@ -933,9 +705,11 @@ along with versions of each sub package:
Output Key Type Description
Total packages validatedInteger total dependency packages under the said metapackage
installed dependency packages but with wrong versions
+ The check will emit a result message with the following format: - Meta package >metapakcage-name> : + + Meta package : Package installed version is Package installed version is Package installed version is @@ -945,64 +719,129 @@ The check will emit a result message with the following format: Missing packages : <0> Version mismatch packages : <0> -#### Examples +#### Examples + +**Example:** + +Run: -**Example 1:** + ./rvs -c conf/rcqt_single.conf -In this example, given package has all dependencies installed. +Configuration (`conf/rcqt_single.conf`, first action): actions: - name: metapackage-validation + device: all module: rcqt - package: rocm-ml-sdk - -The output for such configuration is: - [RESULT] [3648664.1164 ] Action name :metapackage-validation - [RESULT] [3648664.1363 ] Module name :rcqt - - Meta package rocm-ml-sdk : - Package miopen-hip-dev installed version is 3.3.0.60300 - Package rocm-core installed version is 6.3.0.60300 - Package rocm-hip-sdk installed version is 6.3.0.60300 - Package rocm-ml-libraries installed version is 6.3.0.60300 + package: rocm rocm-developer-tools rocm-openmp rocm-opencl-sdk rocm-hip + +Sample output (first action, abridged): + + [RESULT] [302695.777282] Action name :metapackage-validation + [RESULT] [302695.777442] Module name :rcqt + Meta package rocm : + json log file is /var/tmp/rvs_1784325458865.json + Package half installed version is 1.12.0.70201 + Package migraphx installed version is 2.15.0.70201 + Package migraphx-dev installed version is 2.15.0.70201 + Package miopen-hip installed version is 3.5.1.70201 + Package miopen-hip-dev installed version is 3.5.1.70201 + Package mivisionx installed version is 3.5.0.70201 + Package mivisionx-dev installed version is 3.5.0.70201 + Package rocm-cmake installed version is 0.14.0.70201 + Package rocm-core installed version is 7.2.1.70201 + Package rocm-developer-tools installed version is 7.2.1.70201 + Package rocm-hip installed version is 7.2.1.70201 + Package rocm-llvm installed version is 22.0.0.26084.70201 + Package rocm-opencl-sdk installed version is 7.2.1.70201 + Package rocm-openmp installed version is 7.2.1.70201 + Package rocminfo installed version is 1.0.0.70201 + Package rpp installed version is 2.2.1.70201 + Package rpp-dev installed version is 2.2.1.70201 Meta package validation complete : - Total packages validated : 4 - Installed packages : 4 + Total packages validated : 17 + Installed packages : 17 + Missing packages : 0 + Version mismatch packages : 0 + ... + Package rocm-smi-lib installed version is 7.8.0.70201 + Package rocminfo installed version is 1.0.0.70201 + Package rocprim-dev installed version is 4.2.0.70201 + Package rocrand installed version is 4.2.0.70201 + Package rocrand-dev installed version is 4.2.0.70201 + Package rocsolver installed version is 3.32.0.70201 + Package rocsolver-dev installed version is 3.32.0.70201 + Package rocsparse installed version is 4.2.0.70201 + Package rocsparse-dev installed version is 4.2.0.70201 + Package rocthrust-dev installed version is 4.2.0.70201 + Package rocwmma-dev installed version is 2.2.0.70201 + Meta package validation complete : + Total packages validated : 52 + Installed packages : 52 Missing packages : 0 Version mismatch packages : 0 - -For other cases, we will see mismatched/missing packages printed -with respective count + +=====================================================================+ + | ROCm Validation Suite (RVS) Summary | + +=====================================================================+ + | System Overview | + +---------------------------------------------------------------------+ + | Operating System | Ubuntu 22.04.5 LTS | + | RVS version | 1.6.75 | + | ROCm version | 7.2.1-81 | + | amdgpu version | 6.16.13 | + | GPUs | 8 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 42583 | AMD Instinct MI355X - 27226 | + | 0 - 2 - 0000:05:00.0 | 1 - 3 - 0000:15:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 36479 | AMD Instinct MI355X - 17010 | + | 2 - 4 - 0000:65:00.0 | 3 - 5 - 0000:75:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 1590 | AMD Instinct MI355X - 51771 | + | 4 - 6 - 0000:85:00.0 | 5 - 7 - 0000:95:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 11806 | AMD Instinct MI355X - 57875 | + | 6 - 8 - 0000:e5:00.0 | 7 - 9 - 0000:f5:00.0 | + +=====================================================================+ + | Action Name | Module | Result | + +=====================================================================+ + | metapackage-validation | RCQT | PASS | + +---------------------------------------------------------------------+ + +For other cases, mismatched or missing packages are printed with respective counts. ### Packages installation check This action checks if the package is installed. - Packages are provided against key **rpmpackagelist** and **debpackagelist** + Packages are provided against key `rpmpackagelist` and `debpackagelist` #### Packages installation Specific Keys Input keys are described in the table below: - - +
+
Config Key Type Description
+ - +
Config Key Type Description
rpmpackagelistCollection of StringsSpecifies the packages checked if installed on system for rhel/centos family.
Specifies the packages checked if installed on system for rhel family.
debpackagelistCollection of Strings Specifies the packages checked if installed on system for ubuntu family.
+ #### Output Output keys are described in the table below: - - +
+
Output Key Type Description
+ - + @@ -1012,16 +851,16 @@ Output keys are described in the table below:
Output Key Type Description
PackageString Name of checked package
versionFloating Number
versionString Installed version of the package
Missing packagesIntegerNumber of packages installed.
+ #### Examples **Example 1:** -In this example, given user does not exist. +In this example, all given packages are installed. actions: - name: packagelist-install-validation - device: all module: rcqt rpmpackagelist: rocm-hip-libraries rocm-core @@ -1037,7 +876,7 @@ The output for such configuration is: Installed packages : 2 -## PEQT Module +## PEQT module PCI Express Qualification Tool module targets and qualifies the configuration of the platforms PCIe connections to the GPUs. The purpose of the PEQT module is to @@ -1049,52 +888,75 @@ control, status and capabilities registers. These registers are specified in the PCI Express Base Specification, Revision 3. Iteration keys, i.e. count, wait and duration will be ignored for actions using the PEQT module. -### Module Specific Keys +```{note} +The PEQT module requires elevated privileges. Run RVS with `sudo` when using this module. +``` + +### Module specific keys Module specific output keys are described in the table below: - - - - -
Config Key Type Description
capabilityCollection of Structures with the -following format:\n{String,String}The PCIe capability key contains a collection of structures that specify -which PCIe capability to check and the expected value of the capability. A check -structure must contain the PCIe capability value, but an expected value may be -omitted. The value of all valid capabilities that are a part of this collection -will be entered into the capability_value field. Possible capabilities, and -their value types are:\n\n -link_cap_max_speed\n -link_cap_max_width\n -link_stat_cur_speed\n -link_stat_neg_width\n -slot_pwr_limit_value\n -slot_physical_num\n -bus_id\n -atomic_op_32_completer\n -atomic_op_64_completer\n -atomic_op_128_CAS_completer\n -atomic_op_routing\n -dev_serial_num\n -kernel_driver\n -pwr_base_pwr\n -pwr_rail_type\n -device_id\n -vendor_id\n\n + +
+ + + + + + + + + + + + + + + +
Config KeyTypeDescription
capability + Collection of Structures with the following format: +
{String, String}
+
+ The PCIe capability key contains a collection of structures that specify + which PCIe capability to check and the expected value of the capability. A check + structure must contain the PCIe capability value, but an expected value may be + omitted. The value of all valid capabilities that are a part of this collection + will be entered into the capability_value field. Possible capabilities, + and their value types are: +
    +
  • link_cap_max_speed
  • +
  • link_cap_max_width
  • +
  • link_stat_cur_speed
  • +
  • link_stat_neg_width
  • +
  • slot_pwr_limit_value
  • +
  • slot_physical_num
  • +
  • bus_id
  • +
  • atomic_op_32_completer
  • +
  • atomic_op_64_completer
  • +
  • atomic_op_128_CAS_completer
  • +
  • atomic_op_routing
  • +
  • dev_serial_num
  • +
  • kernel_driver
  • +
  • device_id
  • +
  • vendor_id
  • +
+
+
The expected value String is a regular expression that is used to check the actual value of the capability. -
- ### Output Module specific output keys are described in the table below: - - + +
+
Output Key Type Description
+
Output Key Type Description
capability_valueCollection of Strings For each of the capabilities specified in the capability key, the actual value of the capability will be returned, represented as a String.
passString 'true' if all of the properties match the values given, 'false' otherwise.
+ The qualification check queries the specified PCIe capabilities and properties and checks that their actual values satisfy the regular expression @@ -1111,12 +973,17 @@ chapters in the PCI Express Base Specification, Revision 3. ### Examples -**Example 1:** +**Example:** + +Run: -A regular PEQT configuration file looks like this: + ./rvs -c conf/peqt_single.conf + +Configuration (`conf/peqt_single.conf`, first action): actions: - name: pcie_act_1 + device: all module: peqt capability: link_cap_max_speed: @@ -1125,7 +992,7 @@ A regular PEQT configuration file looks like this: link_stat_neg_width: slot_pwr_limit_value: slot_physical_num: - device_id: + deviceid: vendor_id: kernel_driver: dev_serial_num: @@ -1137,275 +1004,47 @@ A regular PEQT configuration file looks like this: atomic_op_32_completer: atomic_op_64_completer: atomic_op_128_CAS_completer: - device: all - -Please note: -- when setting the 'device' configuration key to 'all', the RVS will detect all the AMD compatible GPUs and run the test on all of them - -- there are no regular expression for this .conf file, therefore RVS will report TRUE if at least one AMD compatible GPU is registered within the system. Otherwise it will report FALSE. - -Please note that the Power Budgeting capability is a dynamic one, having the following form: - - __ - -where: - - PM_State = D0/D1/D2/D3 - Type=PMEAux/Auxiliary/Idle/Sustained/Maximum - PowerRail = Power_12V/Power_3_3V/Power_1_5V_1_8V/Thermal - -When the RVS tool runs against such a configuration file, it will query for the -all the PCIe capabilities specified under the capability list (and log the -corresponding values) for all the AMD compatible GPUs. For those PCIe -capabilities that are not supported by the HW platform were the RVS is running, -a "NOT SUPPORTED" message will be logged. - -The output for such a configuration file may look like this: - - - [INFO ] [177628.401176] pcie_act_1 peqt D0_Maximum_Power_12V NOT SUPPORTED - [INFO ] [177628.401229] pcie_act_1 peqt D0_Maximum_Power_3_3V NOT SUPPORTED - [INFO ] [177628.401248] pcie_act_1 peqt D0_Sustained_Power_12V NOT SUPPORTED - [INFO ] [177628.401269] pcie_act_1 peqt D0_Sustained_Power_3_3V NOT SUPPORTED - [INFO ] [177628.401282] pcie_act_1 peqt atomic_op_128_CAS_completer FALSE - [INFO ] [177628.401291] pcie_act_1 peqt atomic_op_32_completer FALSE - [INFO ] [177628.401303] pcie_act_1 peqt atomic_op_64_completer FALSE - [INFO ] [177628.401311] pcie_act_1 peqt atomic_op_routing TRUE - [INFO ] [177628.401317] pcie_act_1 peqt dev_serial_num NOT SUPPORTED - [INFO ] [177628.401323] pcie_act_1 peqt device_id 26720 - [INFO ] [177628.401334] pcie_act_1 peqt kernel_driver amdgpu - [INFO ] [177628.401342] pcie_act_1 peqt link_cap_max_speed 8 GT/s - [INFO ] [177628.401352] pcie_act_1 peqt link_cap_max_width x16 - [INFO ] [177628.401359] pcie_act_1 peqt link_stat_cur_speed 8 GT/s - [INFO ] [177628.401367] pcie_act_1 peqt link_stat_neg_width x16 - [INFO ] [177628.401375] pcie_act_1 peqt slot_physical_num #0 - [INFO ] [177628.401396] pcie_act_1 peqt slot_pwr_limit_value 0.000W - [INFO ] [177628.401402] pcie_act_1 peqt vendor_id 4098 - [INFO ] [177628.401656] pcie_act_1 peqt D0_Maximum_Power_12V NOT SUPPORTED - [INFO ] [177628.401675] pcie_act_1 peqt D0_Maximum_Power_3_3V NOT SUPPORTED - [INFO ] [177628.401692] pcie_act_1 peqt D0_Sustained_Power_12V NOT SUPPORTED - [INFO ] [177628.401709] pcie_act_1 peqt D0_Sustained_Power_3_3V NOT SUPPORTED - [INFO ] [177628.401719] pcie_act_1 peqt atomic_op_128_CAS_completer FALSE - [INFO ] [177628.401728] pcie_act_1 peqt atomic_op_32_completer FALSE - [INFO ] [177628.401736] pcie_act_1 peqt atomic_op_64_completer FALSE - [INFO ] [177628.401745] pcie_act_1 peqt atomic_op_routing TRUE - [INFO ] [177628.401750] pcie_act_1 peqt dev_serial_num NOT SUPPORTED - [INFO ] [177628.401757] pcie_act_1 peqt device_id 26720 - [INFO ] [177628.401771] pcie_act_1 peqt kernel_driver amdgpu - [INFO ] [177628.401781] pcie_act_1 peqt link_cap_max_speed 8 GT/s - [INFO ] [177628.401788] pcie_act_1 peqt link_cap_max_width x16 - [INFO ] [177628.401794] pcie_act_1 peqt link_stat_cur_speed 8 GT/s - [INFO ] [177628.401800] pcie_act_1 peqt link_stat_neg_width x16 - [INFO ] [177628.401806] pcie_act_1 peqt slot_physical_num #0 - [INFO ] [177628.401814] pcie_act_1 peqt slot_pwr_limit_value 0.000W - [INFO ] [177628.401819] pcie_act_1 peqt vendor_id 4098 - [RESULT] [177628.403781] pcie_act_1 peqt TRUE - -**Example 2:** - -Another example of a configuration file, which queries for a smaller subset of PCIe capabilities but adds regular expressions check, is given below - - actions: - - name: pcie_act_1 - module: peqt - capability: - link_cap_max_speed: '^(2\.5 GT\/s|5 GT\/s|8 GT\/s)$' - link_cap_max_width: - link_stat_cur_speed: '^(2\.5 GT\/s|5 GT\/s|8 GT\/s)$' - link_stat_neg_width: - slot_pwr_limit_value: '[a-b][d-' - slot_physical_num: - device_id: - vendor_id: - kernel_driver: - device: all - -For this example, the expected PEQT check result is TRUE if: - -- at least one AMD compatible GPU is registered within the system and: -- all \ values for all AMD compatible GPUs match the given regular expression and -- all \ values for all AMD compatible GPUs match the given regular expression - -Please note that the \ regular expression is not valid and -will be skipped without affecting the PEQT module's check RESULT (however, an -error will be logged out) - -**Example 3:** - -Another example with even more regular expressions is given below. The expected -PEQT check result is TRUE if at least one AMD compatible GPU having the ID 3254 -or 33367 is registered within the system and all the PCIe capabilities values -match their corresponding regular expressions. - - actions: - - name: pcie_act_1 - module: peqt - deviceid: 26720 - capability: - link_cap_max_speed: '^(2\.5 GT\/s|5 GT\/s|8 GT\/s)$' - link_cap_max_width: ^(x8|x16)$ - link_stat_cur_speed: '^(8 GT\/s)$' - link_stat_neg_width: ^(x8|x16)$ - kernel_driver: ^amdgpu$ - atomic_op_routing: ^((TRUE|FALSE){1})$ - atomic_op_32_completer: ^((TRUE|FALSE){1})$ - atomic_op_64_completer: ^((TRUE|FALSE){1})$ - atomic_op_128_CAS_completer: ^((TRUE|FALSE){1})$ - device: 3254 33367 - -## SMQT Module -The GPU SBIOS mapping qualification tool is designed to verify that a platform’s -SBIOS has satisfied the BAR mapping requirements for VDI and Radeon Instinct -products for ROCm support. These are the current BAR requirements:\n\n - -BAR 1: GPU Frame Buffer BAR – In this example it happens to be 256M, but -typically this will be size of the GPU memory (typically 4GB+). This BAR has to -be placed < 2^40 to allow peer- to-peer access from other GFX8 AMD GPUs. For -GFX9 (Vega GPU) the BAR has to be placed < 2^44 to allow peer-to-peer access -from other GFX9 AMD GPUs.\n\n - -BAR 2: Doorbell BAR – The size of the BAR is typically will be < 10MB (currently -fixed at 2MB) for this generation GPUs. This BAR has to be placed < 2^40 to -allow peer-to-peer access from other current generation AMD GPUs.\n\n -BAR 3: IO BAR - This is for legacy VGA and boot device support, but since this -the GPUs in this project are not VGA devices (headless), this is not a concern -even if the SBIOS does not setup.\n\n - -BAR 4: MMIO BAR – This is required for the AMD Driver SW to access the -configuration registers. Since the reminder of the BAR available is only 1 DWORD -(32bit), this is placed < 4GB. This is fixed at 256KB.\n\n - -BAR 5: Expansion ROM – This is required for the AMD Driver SW to access the -GPU’s video-BIOS. This is currently fixed at 128KB.\n\n - -Refer to the ROCm Use of Advanced PCIe Features and Overview of How BAR Memory -is Used In ROCm Enabled System web page for more information about how BAR -memory is initialized by VDI and Radeon products. Iteration keys, i.e. count, -wait and duration will be ignored. - -### Module Specific Keys - -Module specific output keys are described in the table below: - - - - - - - - - - - - - - - - - - - - - - -
Config Key Type Description
bar1_req_sizeIntegerThis is an integer specifying the required size of the BAR1 frame buffer -region.
bar1_base_addr_minIntegerThis is an integer specifying the minimum value the BAR1 base address can -be.
bar1_base_addr_maxIntegerThis is an integer specifying the maximum value the BAR1 base address can -be.
bar2_req_sizeIntegerThis is an integer specifying the required size of the BAR2 frame buffer -region.
bar2_base_addr_minIntegerThis is an integer specifying the minimum value the BAR2 base address can -be.
bar2_base_addr_maxIntegerThis is an integer specifying the maximum value the BAR2 base address can -be.
bar4_req_sizeIntegerThis is an integer specifying the required size of the BAR4 frame buffer -region.
bar4_base_addr_minIntegerThis is an integer specifying the minimum value the BAR4 base address can -be.
bar4_base_addr_maxIntegerThis is an integer specifying the maximum value the BAR4 base address can -be.
bar5_req_sizeIntegerThis is an integer specifying the required size of the BAR5 frame buffer -region.
- -### Output - -Module specific output keys are described in the table below: - - - - - - - - - - -
Output Key Type Description
bar1_sizeIntegerThe actual size of BAR1.
bar1_base_addrIntegerThe actual base address of BAR1 -memory.
bar2_sizeIntegerThe actual size of BAR2.
bar2_base_addrIntegerThe actual base address of BAR2 -memory.
bar4_sizeIntegerThe actual size of BAR4.
bar4_base_addrIntegerThe actual base address of BAR4 -memory.
bar5_sizeIntegerThe actual size of BAR5.
passString 'true' if all of the properties match the -values given, 'false' otherwise.
- -The qualification check will query the specified bar properties and check that -they satisfy the give parameters. The pass output key will be true and the test -will pass if all of the BAR properties satisfy the constraints. After the check -is finished, the following informational messages will be generated: - - [INFO ][][] smqt bar1_size - [INFO ][][] smqt bar1_base_addr - [INFO ][][] smqt bar2_size - [INFO ][][] smqt bar2_base_addr - [INFO ][][] smqt bar4_size - [INFO ][][] smqt bar4_base_addr - [INFO ][][] smqt bar5_size - [RESULT][][] smqt - - -### Examples - -**Example 1:** -Consider this file (sizes are in bytes): - - actions: - - name: action_1 - device: all - module: smqt - bar1_req_size: 17179869184 - bar1_base_addr_min: 0 - bar1_base_addr_max: 17592168044416 - bar2_req_size: 2097152 - bar2_base_addr_min: 0 - bar2_base_addr_max: 1099511627776 - bar4_req_size: 262144 - bar4_base_addr_min: 0 - bar4_base_addr_max: 17592168044416 - bar5_req_size: 131072 - -Results for three GPUs are: - - [INFO ] [257936.568768] [action_1] smqt bar1_size 17179869184 (16.00 GB) - [INFO ] [257936.568768] [action_1] smqt bar1_base_addr 13C0000000C - [INFO ] [257936.568768] [action_1] smqt bar2_size 2097152 (2.00 MB) - [INFO ] [257936.568768] [action_1] smqt bar2_base_addr 13B0000000C - [INFO ] [257936.568768] [action_1] smqt bar4_size 524288 (512.00 KB) - [INFO ] [257936.568768] [action_1] smqt bar4_base_addr E4B00000 - [INFO ] [257936.568768] [action_1] smqt bar5_size 0 (0.00 B) - [RESULT] [257936.568920] [action_1] smqt fail - [INFO ] [257936.569234] [action_1] smqt bar1_size 17179869184 (16.00 GB) - [INFO ] [257936.569234] [action_1] smqt bar1_base_addr 1A00000000C - [INFO ] [257936.569234] [action_1] smqt bar2_size 2097152 (2.00 MB) - [INFO ] [257936.569234] [action_1] smqt bar2_base_addr 19F0000000C - [INFO ] [257936.569234] [action_1] smqt bar4_size 524288 (512.00 KB) - [INFO ] [257936.569234] [action_1] smqt bar4_base_addr E9900000 - [INFO ] [257936.569234] [action_1] smqt bar5_size 0 (0.00 B) - [RESULT] [257936.569281] [action_1] smqt fail - [INFO ] [257936.570798] [action_1] smqt bar1_size 17179869184 (16.00 GB) - [INFO ] [257936.570798] [action_1] smqt bar1_base_addr 16C0000000C - [INFO ] [257936.570798] [action_1] smqt bar2_size 2097152 (2.00 MB) - [INFO ] [257936.570798] [action_1] smqt bar2_base_addr 1710000000C - [INFO ] [257936.570798] [action_1] smqt bar4_size 524288 (512.00 KB) - [INFO ] [257936.570798] [action_1] smqt bar4_base_addr E7300000 - [INFO ] [257936.570798] [action_1] smqt bar5_size 0 (0.00 B) - [RESULT] [257936.570837] [action_1] smqt fail - -In this example, BAR sizes reported by GPUs match those listed in configuration -key except for the BAR5, hence the test fails. - -## PBQT Module +```{note} +- When setting the `device` configuration key to `all`, the RVS will detect all the AMD compatible GPUs and run the test on all of them. +- With no regular expressions specified, RVS reports `TRUE` if at least one AMD compatible GPU is registered within the system. Otherwise it reports `FALSE`. +``` + +Sample output (first action, abridged): + + [RESULT] [302695.130515] Action name :pcie_act_1 + [RESULT] [302695.251177] Module name :peqt + [RESULT] [302695.289685] [pcie_act_1] peqt true + + +=====================================================================+ + | ROCm Validation Suite (RVS) Summary | + +=====================================================================+ + | System Overview | + +---------------------------------------------------------------------+ + | Operating System | Ubuntu 22.04.5 LTS | + | RVS version | 1.6.75 | + | ROCm version | 7.2.1-81 | + | amdgpu version | 6.16.13 | + | GPUs | 8 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 42583 | AMD Instinct MI355X - 27226 | + | 0 - 2 - 0000:05:00.0 | 1 - 3 - 0000:15:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 36479 | AMD Instinct MI355X - 17010 | + | 2 - 4 - 0000:65:00.0 | 3 - 5 - 0000:75:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 1590 | AMD Instinct MI355X - 51771 | + | 4 - 6 - 0000:85:00.0 | 5 - 7 - 0000:95:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 11806 | AMD Instinct MI355X - 57875 | + | 6 - 8 - 0000:e5:00.0 | 7 - 9 - 0000:f5:00.0 | + +=====================================================================+ + | Action Name | Module | Result | + +=====================================================================+ + | pcie_act_1 | PEQT | PASS | + +---------------------------------------------------------------------+ + +## PBQT module The P2P Qualification Tool is designed to provide the list of all GPUs that support P2P and characterize the P2P links between peers. In addition to testing @@ -1414,10 +1053,11 @@ between all unique P2P pairs for performance evaluation. These are known as device-to-device transfers, and can be either uni-directional or bi-directional. The average bandwidth obtained is reported to help debug low bandwidth issues. -### Module Specific Keys +### Module specific keys - - +
+
Config Key Type Description
+ to a specific device type corresponding to the deviceid. +devices pass the P2P check. Note: setting this to false currently causes the +module to exit with an error in the current implementation; bandwidth testing +must be enabled for a successful run. + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Config Key Type Description
peersCollection of Strings This is a required key, and specifies the set of GPU(s) considered being peers of the GPU specified in the action. If ‘all’ is specified, all other @@ -1428,7 +1068,9 @@ specified in the list will be considered.
test_bandwidthBool If this key is set to true the P2P bandwidth benchmark will run if a pair of -devices pass the P2P check.
bidirectionalBool This option is only used if test_bandwidth key is true. This specifies the type of transfer to run:\n @@ -1490,17 +1132,87 @@ key will be performed.
This is a positive integer indicating type of link to be included in bandwidth test. Numbering follows that listed in **hsa\_amd\_link\_info\_type\_t** in **hsa\_ext\_amd.h** file.
b2bBoolIf true, transfers are run back-to-back continuously for the full test +duration. Only applicable when using the native transfer method. If not +specified the default value is false.
hot_callsIntegerNumber of timed transfer iterations used to measure bandwidth. If not +specified the default value is 1.
warm_callsIntegerNumber of warm-up transfer iterations run before timing begins. Warm-up +iterations are not included in the bandwidth measurement. If not specified +the default value is 1.
transfer_methodStringTransfer backend to use. Accepted values: +native – Use the ROCm native SDMA transfer path (default). +transferbench – Use the TransferBench library as the transfer backend. +If not specified the default is native.
transferbench_testStringTransferBench test type. Only applicable when +transfer_method: transferbench. Accepted values: +p2p – Point-to-point transfer test (default). +alltoall – All-to-all transfer test across all participating GPUs. +If not specified the default is p2p.
executorStringExecution engine used by TransferBench. Only applicable when +transfer_method: transferbench. Accepted values: +gfx – Use GPU shader kernels (default). +dma – Use the DMA engine. +If not specified the default is gfx.
subexecutorIntegerNumber of sub-executors (wavefronts or threads) per transfer. Only +applicable when transfer_method: transferbench. If not specified the +default value is 1.
gfx_unrollIntegerUnroll factor for the GFX shader kernel. Only applicable when +transfer_method: transferbench and executor: gfx. If not +specified the default value is 4.
use_remote_readIntegerIf set to 1, transfers use remote read instead of remote write. If not +specified the default value is 0 (remote write).
a2a_modeIntegerAll-to-all transfer pattern mode. Only applicable when +transferbench_test: alltoall. If not specified the default value is +0.
a2a_directIntegerIf set to 1, enables direct peer-to-peer transfers in the all-to-all +test. Only applicable when transferbench_test: alltoall. If not +specified the default value is 1.
a2a_localIntegerIf set to 1, includes local (same-GPU) transfers in the all-to-all test. +Only applicable when transferbench_test: alltoall. If not specified +the default value is 0.
a2a_num_gpusIntegerNumber of GPUs to include in the all-to-all test. A value of 0 means all +detected GPUs are included. Only applicable when +transferbench_test: alltoall. If not specified the default value is +0.
+ -Please note that suitable values for **log\_interval** and **duration** depend +Suitable values for **log\_interval** and **duration** depend on your system. -- **log_interval**, in sequential mode, should be long enough to allow all -transfer tests to finish at lest once or "(pending)" and "(*)" will be displayed +- `log_interval`, in sequential mode, should be long enough to allow all +transfer tests to finish at least once or "(pending)" and "(*)" will be displayed (see below). Number of transfers depends on number of peer NUMA nodes in your system. In parallel mode, it should be roughly 1.5 times the duration of single longest individual test. -- **duration**, regardless of mode should be at least, 4 * log_interval. +- `duration`, regardless of mode should be at least, 4 * `log_interval`. You may obtain indication of how long single transfer between two NUMA nodes take by running test with "-d 4" switch and observing DEBUG messages for @@ -1520,13 +1232,15 @@ transfer start/finish. An output may look like this: [DEBUG ] [183944.700868] [action_1] pbqt transfer 6 4 finish From this printout, it can be concluded that single transfer takes on average -800ms. Values for **log\_interval** and **duration** should be set accordingly. +800ms. Values for `log_interval` and `duration` should be set accordingly. ### Output Module specific output keys are described in the table below: - - + +
+
Output Key Type Description
+
Output Key Type Description
p2p_resultBool Indicates if the gpu and the specified peer have P2P capabilities. If this quantity is true, the GPU pair tested has p2p capabilities. If false, they are @@ -1563,222 +1277,117 @@ peers. You may need to increase test duration.
durationFloat Cumulative duration of all transfers between the two particular nodes
+ -If the value of test_bandwidth key is false, the tool will only try to determine -if the GPU(s) in the peers key are P2P to the action’s GPU. In this case the -bidirectional and log_interval values will be ignored, if they are specified. If -a gpu is a P2P peer to the device the test will pass, otherwise it will fail. A -message indicating the result will be provided for each GPUs specified. It will -have the following format: +The P2P capability check logs a result message for each device-peer pair. +Self-pairs (same GPU) are not logged. The message format is: - [RESULT][][] p2p peers: distance: :[ :] + [] p2p [GPU:: - - ] [GPU:: - - ] peers: distance: : -If the value of test_bandwidth is true bandwidth testing between the device and -each of its peers will take place in parallel or in sequence, depending on the -value of the parallel flag. During the duration of bandwidth benchmarking, -informational output providing the moving average of the transfer’s bandwidth -will be calculated and logged at every time increment specified by the -log_interval parameter. The messages will have the following output: +If `test_bandwidth` is true, bandwidth testing between the device and each of +its peers will take place in parallel or in sequence, depending on the value of +the `parallel` flag. During bandwidth benchmarking, informational output +providing the moving average of the transfer's bandwidth is logged at every +`log_interval`: - [INFO ][][] p2p-bandwidth [] bidirectional: + [] p2p-bandwidth[] [GPU:: - - ] [GPU:: - - ] bidirectional: GBps -At the end of the test the average bytes/second will be calculated over the -entire test duration, and will be logged as a result: +At the end of the test, the average bandwidth over the entire test duration is +logged as a result: - [RESULT][][] p2p-bandwidth [] bidirectional: + [] p2p-bandwidth[] [GPU:: - - ] [GPU:: - - ] bidirectional: GBps duration: secs +When `transferbench_test: alltoall` is used, additional aggregate output lines +are emitted: -### Examples - - -**Example 1:** - -Here all source GPUs (device: all) with all destination GPUs (peers: all) are -tested for p2p capability with no bandwidth testing (test_bandwidth: false). - - actions: - - name: action_1 - device: all - module: pbqt - peers: all - test_bandwidth: false - + [] a2a-p2p-bandwidth[] ... + [] a2a-gpu-bandwidth ... Aggregate peer bandwidth: GBps + [] a2a-bandwidth [ GPUs][ p2p transfers] Aggregate bandwidth: GBps -Possible result is: +### Examples - [RESULT] [1656631.262875] [action_1] p2p 3254 3254 peers:false distance:-1 - [RESULT] [1656631.262968] [action_1] p2p 3254 50599 peers:true distance:56 HyperTransport:56 - [RESULT] [1656631.263039] [action_1] p2p 3254 33367 peers:true distance:56 HyperTransport:56 - [RESULT] [1656631.263103] [action_1] p2p 50599 3254 peers:true distance:56 HyperTransport:56 - [RESULT] [1656631.263151] [action_1] p2p 50599 50599 peers:false distance:-1 - [RESULT] [1656631.263203] [action_1] p2p 50599 33367 peers:true distance:56 HyperTransport:56 - [RESULT] [1656631.263265] [action_1] p2p 33367 3254 peers:true distance:56 HyperTransport:56 - [RESULT] [1656631.263321] [action_1] p2p 33367 50599 peers:true distance:56 HyperTransport:56 - [RESULT] [1656631.263360] [action_1] p2p 33367 33367 peers:false distance:-1 +**Example:** -From the first line of result, we can see that GPU (ID 3254) can't access itself. -From the second line of result, we can see that source GPU (ID 3254) can access destination GPU (ID 50599). +Run: -**Example 2:** + ./rvs -c conf/MI355X/pbqt_single.conf -Here all source GPUs (device: all) with all destination GPUs (peers: all) are -tested for p2p capability including bandwidth testing (test_bandwidth: true) -with bidirectional transfers (bidirectional: true) and with emmediate output -for each completed transfer (log_interval: 0) +Configuration (`conf/MI355X/pbqt_single.conf`, first action): actions: - - name: action_1 + - name: xgmi_d2d_unidir_bandwidth device: all module: pbqt - log_interval: 0 - duration: 0 + log_interval: 5000 + duration: 30000 peers: all test_bandwidth: true - bidirectional: true - -When run with "-d 3" switch, possible result is: - - [RESULT] [1657122.364752] [action_1] p2p 3254 3254 peers:false distance:-1 - [RESULT] [1657122.364845] [action_1] p2p 3254 50599 peers:true distance:56 HyperTransport:56 - [RESULT] [1657122.364917] [action_1] p2p 3254 33367 peers:true distance:56 HyperTransport:56 - [RESULT] [1657122.364985] [action_1] p2p 50599 3254 peers:true distance:56 HyperTransport:56 - [RESULT] [1657122.365037] [action_1] p2p 50599 50599 peers:false distance:-1 - [RESULT] [1657122.365094] [action_1] p2p 50599 33367 peers:true distance:56 HyperTransport:56 - [RESULT] [1657122.365157] [action_1] p2p 33367 3254 peers:true distance:56 HyperTransport:56 - [RESULT] [1657122.365221] [action_1] p2p 33367 50599 peers:true distance:56 HyperTransport:56 - [RESULT] [1657122.365270] [action_1] p2p 33367 33367 peers:false distance:-1 - [INFO ] [1657123.644203] [action_1] p2p-bandwidth [1/6] 3254 50599 bidirectional: true 7.013 GBps - [INFO ] [1657123.644376] [action_1] p2p-bandwidth [2/6] 3254 33367 bidirectional: true 6.615 GBps - [INFO ] [1657123.644453] [action_1] p2p-bandwidth [3/6] 50599 3254 bidirectional: true 2.367 GBps - [INFO ] [1657123.644522] [action_1] p2p-bandwidth [4/6] 50599 33367 bidirectional: true 7.504 GBps - [INFO ] [1657123.644590] [action_1] p2p-bandwidth [5/6] 33367 3254 bidirectional: true 8.207 GBps - [INFO ] [1657123.644673] [action_1] p2p-bandwidth [6/6] 33367 50599 bidirectional: true 7.680 GBps - [INFO ] [1657124.926221] [action_1] p2p-bandwidth [1/6] 3254 50599 bidirectional: true 6.646 GBps - [INFO ] [1657124.926368] [action_1] p2p-bandwidth [2/6] 3254 33367 bidirectional: true 8.418 GBps - [INFO ] [1657124.926438] [action_1] p2p-bandwidth [3/6] 50599 3254 bidirectional: true 7.402 GBps - [INFO ] [1657124.926506] [action_1] p2p-bandwidth [4/6] 50599 33367 bidirectional: true 6.161 GBps - [INFO ] [1657124.926573] [action_1] p2p-bandwidth [5/6] 33367 3254 bidirectional: true 9.024 GBps - [INFO ] [1657124.926640] [action_1] p2p-bandwidth [6/6] 33367 50599 bidirectional: true 8.740 GBps - [INFO ] [1657126.208742] [action_1] p2p-bandwidth [1/6] 3254 50599 bidirectional: true 5.680 GBps - [INFO ] [1657126.208905] [action_1] p2p-bandwidth [2/6] 3254 33367 bidirectional: true 8.011 GBps - [INFO ] [1657126.208990] [action_1] p2p-bandwidth [3/6] 50599 3254 bidirectional: true 3.918 GBps - [INFO ] [1657126.209066] [action_1] p2p-bandwidth [4/6] 50599 33367 bidirectional: true 6.058 GBps - [INFO ] [1657126.209140] [action_1] p2p-bandwidth [5/6] 33367 3254 bidirectional: true 6.650 GBps - [INFO ] [1657126.209213] [action_1] p2p-bandwidth [6/6] 33367 50599 bidirectional: true 0.000 GBps - [RESULT] [1657126.742128] [action_1] p2p-bandwidth [1/6] 3254 50599 bidirectional: true 5.767 GBps duration: 0.368453 sec - [RESULT] [1657126.743287] [action_1] p2p-bandwidth [2/6] 3254 33367 bidirectional: true 6.013 GBps duration: 0.498944 sec - [RESULT] [1657126.744411] [action_1] p2p-bandwidth [3/6] 50599 3254 bidirectional: true 5.278 GBps duration: 0.380393 sec - [RESULT] [1657126.745534] [action_1] p2p-bandwidth [4/6] 50599 33367 bidirectional: true 4.160 GBps duration: 0.484577 sec - [RESULT] [1657126.746684] [action_1] p2p-bandwidth [5/6] 33367 3254 bidirectional: true 5.219 GBps duration: 0.407190 sec - [RESULT] [1657126.747827] [action_1] p2p-bandwidth [6/6] 33367 50599 bidirectional: true 4.001 GBps duration: 0.562350 sec - -We can see that on this particular machine there are three GPUs and six -possible device-to-peer transfers. - -**Example 3:** - -Here some source GPUs (device: 50599) are targeting some destination GPUs -(peers: 33367 3254) with specified log interval (log_interval: 1000) and duration -(duration: 5000). Bandwidth is tested (test_bandwidth: true) but only -unidirectional (bidirectional: false) without parallel execution (parallel: -false). - - actions: - - name: action_1 - device: 50599 - module: pbqt - log_interval: 1000 - duration: 5000 - count: 0 - peers: 33367 3254 - test_bandwidth: true bidirectional: false parallel: false - -Possible output is: - - [RESULT] [1657218.801555] [action_1] p2p 50599 3254 peers:true distance:56 HyperTransport:56 - [RESULT] [1657218.801655] [action_1] p2p 50599 33367 peers:true distance:56 HyperTransport:56 - [INFO ] [1657219.871532] [action_1] p2p-bandwidth [1/2] 50599 3254 bidirectional: false 4.517 GBps - [INFO ] [1657219.871717] [action_1] p2p-bandwidth [2/2] 50599 33367 bidirectional: false 4.475 GBps - [INFO ] [1657220.940263] [action_1] p2p-bandwidth [1/2] 50599 3254 bidirectional: false 4.476 GBps - [INFO ] [1657220.940461] [action_1] p2p-bandwidth [2/2] 50599 33367 bidirectional: false 4.601 GBps - [INFO ] [1657222.7589 ] [action_1] p2p-bandwidth [1/2] 50599 3254 bidirectional: false 4.488 GBps - [INFO ] [1657222.7760 ] [action_1] p2p-bandwidth [2/2] 50599 33367 bidirectional: false 4.470 GBps - [INFO ] [1657223.74647 ] [action_1] p2p-bandwidth [1/2] 50599 3254 bidirectional: false 4.666 GBps - [INFO ] [1657223.74810 ] [action_1] p2p-bandwidth [2/2] 50599 33367 bidirectional: false 4.576 GBps - [RESULT] [1657224.181106] [action_1] p2p-bandwidth [1/2] 50599 3254 bidirectional: false 4.539 GBps duration: 1.321909 sec - [RESULT] [1657224.182255] [action_1] p2p-bandwidth [2/2] 50599 33367 bidirectional: false 4.551 GBps duration: 1.318517 sec - -From the last line of result, we can see that source GPU (ID 50599) can access -destination GPU (ID 33367) and that the bandwidth is 4.495 GBps. - -**Example 4:** - -Here, all GPUs are targeted with bidirectional transfers and parallel execution -of tests: - - actions: - - name: action_1 - device: all - module: pbqt - log_interval: 1200 - duration: 4000 - peers: all - test_bandwidth: true - bidirectional: true - parallel: true - -Possible output is: - - [RESULT] [1657295.937184] [action_1] p2p 3254 3254 peers:false distance:-1 - [RESULT] [1657295.937267] [action_1] p2p 3254 50599 peers:true distance:56 HyperTransport:56 - [RESULT] [1657295.937324] [action_1] p2p 3254 33367 peers:true distance:56 HyperTransport:56 - [RESULT] [1657295.937379] [action_1] p2p 50599 3254 peers:true distance:56 HyperTransport:56 - [RESULT] [1657295.937429] [action_1] p2p 50599 50599 peers:false distance:-1 - [RESULT] [1657295.937482] [action_1] p2p 50599 33367 peers:true distance:56 HyperTransport:56 - [RESULT] [1657295.937543] [action_1] p2p 33367 3254 peers:true distance:56 HyperTransport:56 - [RESULT] [1657295.937607] [action_1] p2p 33367 50599 peers:true distance:56 HyperTransport:56 - [RESULT] [1657295.937655] [action_1] p2p 33367 33367 peers:false distance:-1 - [INFO ] [1657297.216212] [action_1] p2p-bandwidth [1/6] 3254 50599 bidirectional: true 4.972 GBps - [INFO ] [1657297.216351] [action_1] p2p-bandwidth [2/6] 3254 33367 bidirectional: true 8.183 GBps - [INFO ] [1657297.216423] [action_1] p2p-bandwidth [3/6] 50599 3254 bidirectional: true 8.911 GBps - [INFO ] [1657297.216490] [action_1] p2p-bandwidth [4/6] 50599 33367 bidirectional: true 7.690 GBps - [INFO ] [1657297.216558] [action_1] p2p-bandwidth [5/6] 33367 3254 bidirectional: true 7.768 GBps - [INFO ] [1657297.216642] [action_1] p2p-bandwidth [6/6] 33367 50599 bidirectional: true 4.589 GBps - [INFO ] [1657298.487427] [action_1] p2p-bandwidth [1/6] 3254 50599 bidirectional: true 8.778 GBps - [INFO ] [1657298.487593] [action_1] p2p-bandwidth [2/6] 3254 33367 bidirectional: true 7.921 GBps - [INFO ] [1657298.487730] [action_1] p2p-bandwidth [3/6] 50599 3254 bidirectional: true 8.164 GBps - [INFO ] [1657298.487807] [action_1] p2p-bandwidth [4/6] 50599 33367 bidirectional: true 8.921 GBps - [INFO ] [1657298.487878] [action_1] p2p-bandwidth [5/6] 33367 3254 bidirectional: true 8.487 GBps - [INFO ] [1657298.487956] [action_1] p2p-bandwidth [6/6] 33367 50599 bidirectional: true 7.648 GBps - [INFO ] [1657299.760175] [action_1] p2p-bandwidth [1/6] 3254 50599 bidirectional: true 7.210 GBps - [INFO ] [1657299.760249] [action_1] p2p-bandwidth [2/6] 3254 33367 bidirectional: true 4.274 GBps - [INFO ] [1657299.760284] [action_1] p2p-bandwidth [3/6] 50599 3254 bidirectional: true 0.000 GBps - [INFO ] [1657299.760318] [action_1] p2p-bandwidth [4/6] 50599 33367 bidirectional: true 5.942 GBps - [INFO ] [1657299.760349] [action_1] p2p-bandwidth [5/6] 33367 3254 bidirectional: true 0.001 GBps - [INFO ] [1657299.760381] [action_1] p2p-bandwidth [6/6] 33367 50599 bidirectional: true 5.490 GBps - [RESULT] [1657300.293126] [action_1] p2p-bandwidth [1/6] 3254 50599 bidirectional: true 6.964 GBps duration: 0.287248 sec - [RESULT] [1657300.294334] [action_1] p2p-bandwidth [2/6] 3254 33367 bidirectional: true 3.960 GBps duration: 0.536554 sec - [RESULT] [1657300.295528] [action_1] p2p-bandwidth [3/6] 50599 3254 bidirectional: true 5.442 GBps duration: 0.368977 sec - [RESULT] [1657300.296691] [action_1] p2p-bandwidth [4/6] 50599 33367 bidirectional: true 4.187 GBps duration: 0.477756 sec - [RESULT] [1657300.297840] [action_1] p2p-bandwidth [5/6] 33367 3254 bidirectional: true 4.942 GBps duration: 0.607009 sec - [RESULT] [1657300.299016] [action_1] p2p-bandwidth [6/6] 33367 50599 bidirectional: true 3.828 GBps duration: 0.523495 sec - -It can be seen that transfers [2/6] and [5/6] did not take place in the second -log interval so average from the previous cycle is displayed instead and -marked with "(*)" - -## PEBB Module + block_size: 1073741824 + device_id: all + +Sample output (first action, abridged): + + [RESULT] [302380.159634] Action name :xgmi_d2d_unidir_bandwidth + [RESULT] [302380.356991] Module name :pbqt + [RESULT] [302380.357340] [xgmi_d2d_unidir_bandwidth] p2p [GPU:: 2 - 42583 - 0000:05:00.0] [GPU:: 3 - 27226 - 0000:15:00.0] peers:true distance:15 xGMI:15 + [RESULT] [302380.357342] [xgmi_d2d_unidir_bandwidth] p2p [GPU:: 2 - 42583 - 0000:05:00.0] [GPU:: 4 - 36479 - 0000:65:00.0] peers:true distance:15 xGMI:15 + [RESULT] [302380.357344] [xgmi_d2d_unidir_bandwidth] p2p [GPU:: 2 - 42583 - 0000:05:00.0] [GPU:: 5 - 17010 - 0000:75:00.0] peers:true distance:15 xGMI:15 + [RESULT] [302380.357345] [xgmi_d2d_unidir_bandwidth] p2p [GPU:: 2 - 42583 - 0000:05:00.0] [GPU:: 6 - 1590 - 0000:85:00.0] peers:true distance:15 xGMI:15 + [RESULT] [302380.357346] [xgmi_d2d_unidir_bandwidth] p2p [GPU:: 2 - 42583 - 0000:05:00.0] [GPU:: 7 - 51771 - 0000:95:00.0] peers:true distance:15 xGMI:15 + [RESULT] [302380.357347] [xgmi_d2d_unidir_bandwidth] p2p [GPU:: 2 - 42583 - 0000:05:00.0] [GPU:: 8 - 11806 - 0000:e5:00.0] peers:true distance:15 xGMI:15 + [RESULT] [302380.357348] [xgmi_d2d_unidir_bandwidth] p2p [GPU:: 2 - 42583 - 0000:05:00.0] [GPU:: 9 - 57875 - 0000:f5:00.0] peers:true distance:15 xGMI:15 + ... + [RESULT] [302410.443673] [xgmi_d2d_unidir_bandwidth] p2p-bandwidth[48/56] [GPU:: 8 - 11806 - 0000:e5:00.0] [GPU:: 7 - 51771 - 0000:95:00.0] bidirectional: false 61.370 GBps duration: 0.227452 secs + [RESULT] [302410.444726] [xgmi_d2d_unidir_bandwidth] p2p-bandwidth[49/56] [GPU:: 8 - 11806 - 0000:e5:00.0] [GPU:: 9 - 57875 - 0000:f5:00.0] bidirectional: false 61.369 GBps duration: 0.227453 secs + [RESULT] [302410.445779] [xgmi_d2d_unidir_bandwidth] p2p-bandwidth[50/56] [GPU:: 9 - 57875 - 0000:f5:00.0] [GPU:: 2 - 42583 - 0000:05:00.0] bidirectional: false 61.368 GBps duration: 0.227456 secs + [RESULT] [302410.446832] [xgmi_d2d_unidir_bandwidth] p2p-bandwidth[51/56] [GPU:: 9 - 57875 - 0000:f5:00.0] [GPU:: 3 - 27226 - 0000:15:00.0] bidirectional: false 61.373 GBps duration: 0.227441 secs + [RESULT] [302410.447885] [xgmi_d2d_unidir_bandwidth] p2p-bandwidth[52/56] [GPU:: 9 - 57875 - 0000:f5:00.0] [GPU:: 4 - 36479 - 0000:65:00.0] bidirectional: false 59.704 GBps duration: 0.233798 secs + [RESULT] [302410.448939] [xgmi_d2d_unidir_bandwidth] p2p-bandwidth[53/56] [GPU:: 9 - 57875 - 0000:f5:00.0] [GPU:: 5 - 17010 - 0000:75:00.0] bidirectional: false 61.371 GBps duration: 0.227448 secs + [RESULT] [302410.449991] [xgmi_d2d_unidir_bandwidth] p2p-bandwidth[54/56] [GPU:: 9 - 57875 - 0000:f5:00.0] [GPU:: 6 - 1590 - 0000:85:00.0] bidirectional: false 61.367 GBps duration: 0.227463 secs + [RESULT] [302410.451044] [xgmi_d2d_unidir_bandwidth] p2p-bandwidth[55/56] [GPU:: 9 - 57875 - 0000:f5:00.0] [GPU:: 7 - 51771 - 0000:95:00.0] bidirectional: false 61.378 GBps duration: 0.227422 secs + [RESULT] [302410.452097] [xgmi_d2d_unidir_bandwidth] p2p-bandwidth[56/56] [GPU:: 9 - 57875 - 0000:f5:00.0] [GPU:: 8 - 11806 - 0000:e5:00.0] bidirectional: false 61.368 GBps duration: 0.227458 secs + + +=====================================================================+ + | ROCm Validation Suite (RVS) Summary | + +=====================================================================+ + | System Overview | + +---------------------------------------------------------------------+ + | Operating System | Ubuntu 22.04.5 LTS | + | RVS version | 1.6.75 | + | ROCm version | 7.2.1-81 | + | amdgpu version | 6.16.13 | + | GPUs | 8 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 42583 | AMD Instinct MI355X - 27226 | + | 0 - 2 - 0000:05:00.0 | 1 - 3 - 0000:15:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 36479 | AMD Instinct MI355X - 17010 | + | 2 - 4 - 0000:65:00.0 | 3 - 5 - 0000:75:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 1590 | AMD Instinct MI355X - 51771 | + | 4 - 6 - 0000:85:00.0 | 5 - 7 - 0000:95:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 11806 | AMD Instinct MI355X - 57875 | + | 6 - 8 - 0000:e5:00.0 | 7 - 9 - 0000:f5:00.0 | + +=====================================================================+ + | Action Name | Module | Result | + +=====================================================================+ + | xgmi_d2d_unidir_bandwidth | PBQT | PASS | + +---------------------------------------------------------------------+ + +## PEBB module The PCIe Bandwidth Benchmark attempts to saturate the PCIe bus with DMA transfers between system memory and a target GPU card’s memory. These are known as host-to-device or device- to-host transfers, and can be either unidirectional or bidirectional transfers. The maximum bandwidth obtained is reported. -### Module Specific Keys +### Module specific keys - - +
+
Config Key Type Description
+ @@ -1787,26 +1396,25 @@ will be considered. The default value is true. will be considered. The default value is true. - - - - + + + + + + + + + + + + + + + + + + + + + + + + + + +
Config Key Type Description
host_to_deviceBool This key indicates if host to device transfers will be considered. The default value is true.
parallelBoolThis option is only used if the test_bandwidth -key is true.\n -- true – Run all test transfers in parallel.\n -- false – Run test transfers one by one. +Controls whether transfers run in parallel across all CPU–GPU paths or one +by one. +`true` – Run all test transfers in parallel. +`false` – Run test transfers one by one.
durationIntegerThis option is only used if test_bandwidth is true. This key specifies the -duration a transfer test should run, given in milliseconds. If this key is not -specified, the default value is 10000 (10 seconds). +This key specifies the duration a transfer test should run, given in +milliseconds. If this key is not specified, the default value is 10000 (10 +seconds).
log_intervalIntegerThis option is only used if test_bandwidth is true. This is a positive -integer, given in milliseconds, that specifies an interval over which the moving -average of the bandwidth will be calculated and logged. The default value is -1000 (1 second). It must be smaller than the duration key.\n -if this key is 0 (zero), results are displayed as soon as the test transfer +This is a positive integer, given in milliseconds, that specifies an interval +over which the moving average of the bandwidth will be calculated and logged. The +default value is 1000 (1 second). It must be smaller than the duration key. +If this key is 0 (zero), results are displayed as soon as the test transfer is completed.
block_sizeCollection of IntegersOptional. Defines list of block sizes to be used in transfer tests.\n +Optional. Defines list of block sizes to be used in transfer tests. If "all" or missing list of block sizes used in rocm_bandwidth_test is used: - 1 * 1024 - 2 * 1024 @@ -1839,17 +1447,71 @@ key will be performed.
This is a positive integer indicating type of link to be included in bandwidth test. Numbering follows that listed in **hsa\_amd\_link\_info\_type\_t** in **hsa\_ext\_amd.h** file.
b2bBoolIf true, transfers are run back-to-back continuously for the full test +duration. Only applicable when using the native transfer method. If not +specified the default value is false.
hot_callsIntegerNumber of timed transfer iterations used to measure bandwidth. If not +specified the default value is 1.
warm_callsIntegerNumber of warm-up transfer iterations run before timing begins. Warm-up +iterations are not included in the bandwidth measurement. If not specified +the default value is 1.
transfer_methodStringTransfer backend to use. Accepted values: +native – Use the ROCm native SDMA transfer path (default). +transferbench – Use the TransferBench library as the transfer backend. +If not specified the default is native.
executorStringExecution engine used by TransferBench. Only applicable when +transfer_method: transferbench. Accepted values: +gfx – Use GPU shader kernels (default). +dma – Use the DMA engine. +If not specified the default is gfx.
subexecutorIntegerNumber of sub-executors (wavefronts or threads) per transfer. Only +applicable when transfer_method: transferbench. If not specified the +default value is 1.
gfx_unrollIntegerUnroll factor for the GFX shader kernel. Only applicable when +transfer_method: transferbench and executor: gfx. If not +specified the default value is 4.
source_memoryStringMemory type for the transfer source. Only applicable when +transfer_method: transferbench. Accepted values: +cpu – System (host) memory. +gpu – GPU device memory. +null – No explicit memory allocation; let TransferBench decide (default). +If not specified the default is null.
destination_memoryStringMemory type for the transfer destination. Only applicable when +transfer_method: transferbench. Accepted values: +cpu – System (host) memory. +gpu – GPU device memory. +null – No explicit memory allocation; let TransferBench decide (default). +If not specified the default is null.
+ -Please note that suitable values for **log\_interval** and **duration** depend +Suitable values for `log_interval` and `duration` depend on your system. -- **log_interval**, in sequential mode, should be long enough to allow all -transfer tests to finish at lest once or "(pending)" and "(*)" will be displayed +- `log_interval`, in sequential mode, should be long enough to allow all +transfer tests to finish at least once or "(pending)" and "(*)" will be displayed (see below). Number of transfers depends on number of peer NUMA nodes in your system. In parallel mode, it should be roughly 1.5 times the duration of single longest individual test. -- **duration**, regardless of mode should be at least, 4 * log_interval. +- `duration`, regardless of mode should be at least, 4 * `log_interval`. You may obtain indication of how long single transfer between two NUMA nodes take by running test with "-d 4" switch and observing DEBUG messages for @@ -1871,14 +1533,16 @@ transfer start/finish. An output may look like this: [DEBUG ] [187031.605326] [action_1] pebb transfer 0 5 finish From this printout, it can be concluded that single transfer takes on average -5500ms. Values for **log\_interval** and **duration** should be set accordingly. +5500ms. Values for `log_interval` and `duration` should be set accordingly. ### Output Module specific output keys are described in the table below: - - + +
+
Output Key Type Description
+ @@ -1895,8 +1559,8 @@ Module specific output keys are described in the table below: - -
Output Key Type Description
CPU nodeInteger Particular CPU node involved in transfer
distanceInteger
interval_bandwidthFloatThe average bandwidth of a p2p transfer, during the log_interval time -period.\n This field may also take values: +The average bandwidth of a CPU-to-GPU or GPU-to-CPU transfer, during the +log_interval time period.\n This field may also take values: - (pending) - this means that no measurement has taken place yet. - xxxGBps (*) - this means no measurement within current log_interval but @@ -1904,8 +1568,8 @@ average from previous measurements is displayed.
bandwidthFloatThe average bandwidth of a p2p transfer, averaged over the entire test -duration of the interval. This field may also take value: +The average bandwidth of a CPU-to-GPU or GPU-to-CPU transfer, averaged +over the entire test duration. This field may also take value: - (not measured) - this means no test transfer completed for those peers. You may need to increase test duration. @@ -1913,144 +1577,108 @@ peers. You may need to increase test duration.
durationFloat Cumulative duration of all transfers between the two particular nodes
+ -At the beginning, test will display link infor for every CPU/GPU pair: +At the beginning, the test will display link info for every CPU/GPU pair: - [RESULT][][] pcie-bandwidth [] distance: :[ :] + [] pcie-bandwidth [CPU:: ] [GPU:: - - ] distance: : During the execution of the benchmark, informational output providing the moving -average of the bandwidth of the transfer will be calculated and logged. This -interval is provided by the log_interval parameter and will have the following -output format: +average of the bandwidth of the transfer will be calculated and logged at every +`log_interval`: - [INFO ][][] pcie-bandwidth [] h2d: d2h: + [] pcie-bandwidth [] [CPU:: ] [GPU:: - - ] h2d:: d2h:: GBps -At the end of test, the average bytes/second will be calculated over the -entire test duration, and will be logged as a result: +At the end of test, the average bandwidth over the entire test duration is +logged as a result: - [RESULT][][] pcie-bandwidth [] h2d: d2h: + [] pcie-bandwidth [] [CPU:: ] [GPU:: - - ] h2d:: d2h:: GBps duration: secs ### Examples -**Example 1:** +**Example:** -Consider action: +Run: - actions: - - name: action_1 - device: all - module: pebb - log_interval: 0 - duration: 0 - device_to_host: false - host_to_device: true - parallel: false + ./rvs -c conf/MI355X/pebb_single.conf -This will initiate host to device transfer to all GPUs with immediate output -(**parallel: false**, **log_interval: 0**)\n -Output from this action might look like: - - [RESULT] [1658774.978614] [action_1] pcie-bandwidth 0 4 3254 distance:36 HyperTransport:36 - [RESULT] [1658774.978664] [action_1] pcie-bandwidth 1 4 3254 distance:20 PCIe:20 - [RESULT] [1658774.978695] [action_1] pcie-bandwidth 2 4 3254 distance:36 HyperTransport:36 - [RESULT] [1658774.978728] [action_1] pcie-bandwidth 3 4 3254 distance:36 HyperTransport:36 - [RESULT] [1658774.978763] [action_1] pcie-bandwidth 0 5 50599 distance:36 HyperTransport:36 - [RESULT] [1658774.978795] [action_1] pcie-bandwidth 1 5 50599 distance:36 HyperTransport:36 - [RESULT] [1658774.978825] [action_1] pcie-bandwidth 2 5 50599 distance:20 PCIe:20 - [RESULT] [1658774.978856] [action_1] pcie-bandwidth 3 5 50599 distance:36 HyperTransport:36 - [RESULT] [1658774.978889] [action_1] pcie-bandwidth 0 6 33367 distance:36 HyperTransport:36 - [RESULT] [1658774.978922] [action_1] pcie-bandwidth 1 6 33367 distance:36 HyperTransport:36 - [RESULT] [1658774.978952] [action_1] pcie-bandwidth 2 6 33367 distance:36 HyperTransport:36 - [RESULT] [1658774.978982] [action_1] pcie-bandwidth 3 6 33367 distance:20 PCIe:20 - [INFO ] [1658774.983743] [action_1] pcie-bandwidth [1/12] 0 3254 h2d: true d2h: false 12.233 GBps - [INFO ] [1658774.988272] [action_1] pcie-bandwidth [2/12] 1 3254 h2d: true d2h: false 12.227 GBps - [INFO ] [1658774.993197] [action_1] pcie-bandwidth [3/12] 2 3254 h2d: true d2h: false 11.770 GBps - [INFO ] [1658774.998105] [action_1] pcie-bandwidth [4/12] 3 3254 h2d: true d2h: false 11.313 GBps - [INFO ] [1658775.4457 ] [action_1] pcie-bandwidth [5/12] 0 50599 h2d: true d2h: false 12.218 GBps - [INFO ] [1658775.9589 ] [action_1] pcie-bandwidth [6/12] 1 50599 h2d: true d2h: false 10.292 GBps - [INFO ] [1658775.14627 ] [action_1] pcie-bandwidth [7/12] 2 50599 h2d: true d2h: false 10.456 GBps - [INFO ] [1658775.19664 ] [action_1] pcie-bandwidth [8/12] 3 50599 h2d: true d2h: false 10.614 GBps - [INFO ] [1658775.26210 ] [action_1] pcie-bandwidth [9/12] 0 33367 h2d: true d2h: false 12.222 GBps - [INFO ] [1658775.31188 ] [action_1] pcie-bandwidth [10/12] 1 33367 h2d: true d2h: false 12.215 GBps - [INFO ] [1658775.36137 ] [action_1] pcie-bandwidth [11/12] 2 33367 h2d: true d2h: false 12.219 GBps - [INFO ] [1658775.41117 ] [action_1] pcie-bandwidth [12/12] 3 33367 h2d: true d2h: false 12.219 GBps - [RESULT] [1658775.42219 ] [action_1] pcie-bandwidth [1/12] 0 3254 h2d: true d2h: false 12.233 GBps duration: 0.000780 sec - [RESULT] [1658775.42235 ] [action_1] pcie-bandwidth [2/12] 1 3254 h2d: true d2h: false 12.227 GBps duration: 0.000780 sec - [RESULT] [1658775.42246 ] [action_1] pcie-bandwidth [3/12] 2 3254 h2d: true d2h: false 11.770 GBps duration: 0.000810 sec - [RESULT] [1658775.42256 ] [action_1] pcie-bandwidth [4/12] 3 3254 h2d: true d2h: false 11.313 GBps duration: 0.000843 sec - [RESULT] [1658775.42271 ] [action_1] pcie-bandwidth [5/12] 0 50599 h2d: true d2h: false 12.218 GBps duration: 0.000781 sec - [RESULT] [1658775.42286 ] [action_1] pcie-bandwidth [6/12] 1 50599 h2d: true d2h: false 10.292 GBps duration: 0.000927 sec - [RESULT] [1658775.42297 ] [action_1] pcie-bandwidth [7/12] 2 50599 h2d: true d2h: false 10.456 GBps duration: 0.000912 sec - [RESULT] [1658775.42309 ] [action_1] pcie-bandwidth [8/12] 3 50599 h2d: true d2h: false 10.614 GBps duration: 0.000898 sec - [RESULT] [1658775.42321 ] [action_1] pcie-bandwidth [9/12] 0 33367 h2d: true d2h: false 12.222 GBps duration: 0.000780 sec - [RESULT] [1658775.42332 ] [action_1] pcie-bandwidth [10/12] 1 33367 h2d: true d2h: false 12.215 GBps duration: 0.000781 sec - [RESULT] [1658775.42344 ] [action_1] pcie-bandwidth [11/12] 2 33367 h2d: true d2h: false 12.219 GBps duration: 0.000780 sec - [RESULT] [1658775.42355 ] [action_1] pcie-bandwidth [12/12] 3 33367 h2d: true d2h: false 12.219 GBps duration: 0.000780 sec - -**Example 2:** - -Consider action: +Configuration (`conf/MI355X/pebb_single.conf`, first action): actions: - - name: action_1 + - name: pcie_h2d_bandwidth device: all module: pebb - log_interval: 500 - duration: 5000 - device_to_host: true + duration: 30000 + device_to_host: false host_to_device: true - parallel: true - -Here, although parallel execution of transfers is requested, log_interval is to -short for some transfers to complete. For them, cumulative average is displayed -and marked with (*): - - [RESULT] [1659672.517170] [action_1] pcie-bandwidth 0 4 3254 distance:36 HyperTransport:36 - [RESULT] [1659672.517222] [action_1] pcie-bandwidth 1 4 3254 distance:20 PCIe:20 - [RESULT] [1659672.517257] [action_1] pcie-bandwidth 2 4 3254 distance:36 HyperTransport:36 - [RESULT] [1659672.517290] [action_1] pcie-bandwidth 3 4 3254 distance:36 HyperTransport:36 - [RESULT] [1659672.517324] [action_1] pcie-bandwidth 0 5 50599 distance:36 HyperTransport:36 - [RESULT] [1659672.517357] [action_1] pcie-bandwidth 1 5 50599 distance:36 HyperTransport:36 - [RESULT] [1659672.517388] [action_1] pcie-bandwidth 2 5 50599 distance:20 PCIe:20 - [RESULT] [1659672.517419] [action_1] pcie-bandwidth 3 5 50599 distance:36 HyperTransport:36 - [RESULT] [1659672.517452] [action_1] pcie-bandwidth 0 6 33367 distance:36 HyperTransport:36 - [RESULT] [1659672.517483] [action_1] pcie-bandwidth 1 6 33367 distance:36 HyperTransport:36 - [RESULT] [1659672.517515] [action_1] pcie-bandwidth 2 6 33367 distance:36 HyperTransport:36 - [RESULT] [1659672.517546] [action_1] pcie-bandwidth 3 6 33367 distance:20 PCIe:20 - [INFO ] [1659673.49782 ] [action_1] pcie-bandwidth [1/12] 0 3254 h2d: true d2h: true 1.489 GBps - [INFO ] [1659673.49814 ] [action_1] pcie-bandwidth [2/12] 1 3254 h2d: true d2h: true 2.701 GBps - ... - [INFO ] [1659673.582639] [action_1] pcie-bandwidth [1/12] 0 3254 h2d: true d2h: true 1.489 GBps (*) - [INFO ] [1659673.582686] [action_1] pcie-bandwidth [2/12] 1 3254 h2d: true d2h: true 16.367 GBps - [INFO ] [1659673.582700] [action_1] pcie-bandwidth [3/12] 2 3254 h2d: true d2h: true 17.300 GBps - ... - [INFO ] [1659677.851697] [action_1] pcie-bandwidth [1/12] 0 3254 h2d: true d2h: true 16.793 GBps - [INFO ] [1659677.851727] [action_1] pcie-bandwidth [2/12] 1 3254 h2d: true d2h: true 16.872 GBps (*) - [INFO ] [1659677.851741] [action_1] pcie-bandwidth [3/12] 2 3254 h2d: true d2h: true 14.796 GBps (*) - [INFO ] [1659677.851754] [action_1] pcie-bandwidth [4/12] 3 3254 h2d: true d2h: true 20.358 GBps - [INFO ] [1659677.851770] [action_1] pcie-bandwidth [5/12] 0 50599 h2d: true d2h: true 15.632 GBps (*) - [INFO ] [1659677.851828] [action_1] pcie-bandwidth [6/12] 1 50599 h2d: true d2h: true 14.541 GBps (*) + parallel: false + block_size: 1073741824 + link_type: 2 + +Sample output (first action, abridged): + + [RESULT] [302349.492047] Action name :pcie_h2d_bandwidth + [RESULT] [302349.671479] Module name :pebb + [RESULT] [302349.671823] [pcie_h2d_bandwidth] pcie-bandwidth [CPU:: 0] [GPU:: 2 - 42583 - 0000:05:00.0] distance:20 PCIe:20 + [RESULT] [302349.671837] [pcie_h2d_bandwidth] pcie-bandwidth [CPU:: 1] [GPU:: 2 - 42583 - 0000:05:00.0] distance:52 PCIe:52 + [RESULT] [302349.671839] [pcie_h2d_bandwidth] pcie-bandwidth [CPU:: 0] [GPU:: 3 - 27226 - 0000:15:00.0] distance:20 PCIe:20 + [RESULT] [302349.671840] [pcie_h2d_bandwidth] pcie-bandwidth [CPU:: 1] [GPU:: 3 - 27226 - 0000:15:00.0] distance:52 PCIe:52 + [RESULT] [302349.671841] [pcie_h2d_bandwidth] pcie-bandwidth [CPU:: 0] [GPU:: 4 - 36479 - 0000:65:00.0] distance:20 PCIe:20 + [RESULT] [302349.671842] [pcie_h2d_bandwidth] pcie-bandwidth [CPU:: 1] [GPU:: 4 - 36479 - 0000:65:00.0] distance:52 PCIe:52 + [RESULT] [302349.671843] [pcie_h2d_bandwidth] pcie-bandwidth [CPU:: 0] [GPU:: 5 - 17010 - 0000:75:00.0] distance:20 PCIe:20 + [RESULT] [302349.671844] [pcie_h2d_bandwidth] pcie-bandwidth [CPU:: 1] [GPU:: 5 - 17010 - 0000:75:00.0] distance:52 PCIe:52 + [RESULT] [302349.671846] [pcie_h2d_bandwidth] pcie-bandwidth [CPU:: 0] [GPU:: 6 - 1590 - 0000:85:00.0] distance:52 PCIe:52 + [RESULT] [302349.671847] [pcie_h2d_bandwidth] pcie-bandwidth [CPU:: 1] [GPU:: 6 - 1590 - 0000:85:00.0] distance:20 PCIe:20 + [RESULT] [302349.671848] [pcie_h2d_bandwidth] pcie-bandwidth [CPU:: 0] [GPU:: 7 - 51771 - 0000:95:00.0] distance:52 PCIe:52 + [RESULT] [302349.671849] [pcie_h2d_bandwidth] pcie-bandwidth [CPU:: 1] [GPU:: 7 - 51771 - 0000:95:00.0] distance:20 PCIe:20 + [RESULT] [302349.671850] [pcie_h2d_bandwidth] pcie-bandwidth [CPU:: 0] [GPU:: 8 - 11806 - 0000:e5:00.0] distance:52 PCIe:52 + [RESULT] [302349.671851] [pcie_h2d_bandwidth] pcie-bandwidth [CPU:: 1] [GPU:: 8 - 11806 - 0000:e5:00.0] distance:20 PCIe:20 + [RESULT] [302349.671852] [pcie_h2d_bandwidth] pcie-bandwidth [CPU:: 0] [GPU:: 9 - 57875 - 0000:f5:00.0] distance:52 PCIe:52 + [RESULT] [302349.671853] [pcie_h2d_bandwidth] pcie-bandwidth [CPU:: 1] [GPU:: 9 - 57875 - 0000:f5:00.0] distance:20 PCIe:20 + [RESULT] [302379.847423] [pcie_h2d_bandwidth] pcie-bandwidth [ 1/16] [CPU:: 0] [GPU:: 2 - 42583 - 0000:05:00.0] h2d::true d2h::false 57.678 GBps duration: 0.223392 secs + [RESULT] [302379.847438] [pcie_h2d_bandwidth] pcie-bandwidth [ 2/16] [CPU:: 1] [GPU:: 2 - 42583 - 0000:05:00.0] h2d::true d2h::false 54.034 GBps duration: 0.238461 secs ... - [RESULT] [1659678.148280] [action_1] pcie-bandwidth [1/12] 0 3254 h2d: true d2h: true 16.309 GBps duration: 0.061316 sec - [RESULT] [1659678.148318] [action_1] pcie-bandwidth [2/12] 1 3254 h2d: true d2h: true 16.871 GBps duration: 0.118547 sec - [RESULT] [1659678.148332] [action_1] pcie-bandwidth [3/12] 2 3254 h2d: true d2h: true 13.360 GBps duration: 0.149705 sec - [RESULT] [1659678.148349] [action_1] pcie-bandwidth [4/12] 3 3254 h2d: true d2h: true 15.371 GBps duration: 0.130115 sec - [RESULT] [1659678.148363] [action_1] pcie-bandwidth [5/12] 0 50599 h2d: true d2h: true 15.631 GBps duration: 0.127954 sec - [RESULT] [1659678.148377] [action_1] pcie-bandwidth [6/12] 1 50599 h2d: true d2h: true 14.185 GBps duration: 0.140989 sec - [RESULT] [1659678.148390] [action_1] pcie-bandwidth [7/12] 2 50599 h2d: true d2h: true 15.242 GBps duration: 0.131245 sec - [RESULT] [1659678.148404] [action_1] pcie-bandwidth [8/12] 3 50599 h2d: true d2h: true 16.071 GBps duration: 0.124452 sec - [RESULT] [1659678.148418] [action_1] pcie-bandwidth [9/12] 0 33367 h2d: true d2h: true 16.505 GBps duration: 0.121178 sec - [RESULT] [1659678.148432] [action_1] pcie-bandwidth [10/12] 1 33367 h2d: true d2h: true 16.720 GBps duration: 0.059807 sec - [RESULT] [1659678.148445] [action_1] pcie-bandwidth [11/12] 2 33367 h2d: true d2h: true 15.604 GBps duration: 0.128168 sec - [RESULT] [1659678.148458] [action_1] pcie-bandwidth [12/12] 3 33367 h2d: true d2h: true 16.193 GBps duration: 0.123525 sec - -Please note that in link information results, some records could be marked with -(R). This means, that communication is possible if initiated by the destination -NUMA node HSA agent. - -## GST Module + [RESULT] [302379.847449] [pcie_h2d_bandwidth] pcie-bandwidth [ 9/16] [CPU:: 0] [GPU:: 6 - 1590 - 0000:85:00.0] h2d::true d2h::false 57.708 GBps duration: 0.204669 secs + [RESULT] [302379.847451] [pcie_h2d_bandwidth] pcie-bandwidth [10/16] [CPU:: 1] [GPU:: 6 - 1590 - 0000:85:00.0] h2d::true d2h::false 55.223 GBps duration: 0.213882 secs + [RESULT] [302379.847453] [pcie_h2d_bandwidth] pcie-bandwidth [11/16] [CPU:: 0] [GPU:: 7 - 51771 - 0000:95:00.0] h2d::true d2h::false 57.708 GBps duration: 0.204671 secs + [RESULT] [302379.847454] [pcie_h2d_bandwidth] pcie-bandwidth [12/16] [CPU:: 1] [GPU:: 7 - 51771 - 0000:95:00.0] h2d::true d2h::false 57.708 GBps duration: 0.204671 secs + [RESULT] [302379.847472] [pcie_h2d_bandwidth] pcie-bandwidth [13/16] [CPU:: 0] [GPU:: 8 - 11806 - 0000:e5:00.0] h2d::true d2h::false 57.708 GBps duration: 0.204672 secs + [RESULT] [302379.847474] [pcie_h2d_bandwidth] pcie-bandwidth [14/16] [CPU:: 1] [GPU:: 8 - 11806 - 0000:e5:00.0] h2d::true d2h::false 57.708 GBps duration: 0.204670 secs + [RESULT] [302379.847475] [pcie_h2d_bandwidth] pcie-bandwidth [15/16] [CPU:: 0] [GPU:: 9 - 57875 - 0000:f5:00.0] h2d::true d2h::false 57.709 GBps duration: 0.204668 secs + [RESULT] [302379.847476] [pcie_h2d_bandwidth] pcie-bandwidth [16/16] [CPU:: 1] [GPU:: 9 - 57875 - 0000:f5:00.0] h2d::true d2h::false 57.710 GBps duration: 0.204666 secs + + +=====================================================================+ + | ROCm Validation Suite (RVS) Summary | + +=====================================================================+ + | System Overview | + +---------------------------------------------------------------------+ + | Operating System | Ubuntu 22.04.5 LTS | + | RVS version | 1.6.75 | + | ROCm version | 7.2.1-81 | + | amdgpu version | 6.16.13 | + | GPUs | 8 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 42583 | AMD Instinct MI355X - 27226 | + | 0 - 2 - 0000:05:00.0 | 1 - 3 - 0000:15:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 36479 | AMD Instinct MI355X - 17010 | + | 2 - 4 - 0000:65:00.0 | 3 - 5 - 0000:75:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 1590 | AMD Instinct MI355X - 51771 | + | 4 - 6 - 0000:85:00.0 | 5 - 7 - 0000:95:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 11806 | AMD Instinct MI355X - 57875 | + | 6 - 8 - 0000:e5:00.0 | 7 - 9 - 0000:f5:00.0 | + +=====================================================================+ + | Action Name | Module | Result | + +=====================================================================+ + | pcie_h2d_bandwidth | PEBB | PASS | + +---------------------------------------------------------------------+ + +## GST module + The GPU Stress Test drives and measures the specified GPU(s) performance (GFLOPS) - by means of large matrix multiplications using GEMM operation types based computations like SGEMM/DGEMM/HGEMM (Single/Double-precision/Half-precision General Matrix Multiplication) @@ -2062,7 +1690,7 @@ the GFLOPS performance for configured GEMM computation and checks if it meets co performance target. The test passes if it achieves the target performance GFLOPS number during the duration of the test else reported as fail. -This module should be used in conjunction with the GPU Monitor, to watch for +This module should be used in conjunction with GPU monitoring tools (for example, amd-smi), to watch for thermal, power and related anomalies while the target GPU(s) are under realistic load conditions. By setting the appropriate parameters a user can ensure that all GPUs in a node or cluster reach desired performance levels. Further analysis @@ -2070,12 +1698,13 @@ of the generated stats can also show variations in the required power, clocks or temperatures to reach these targets, and thus highlight GPUs or nodes that are operating less efficiently. -### Module Specific Keys +### Module specific keys Module specific keys are described in the table below: - - +
+
Config Key Type Description
+ @@ -2083,7 +1712,7 @@ gigaflops. This parameter is required. - +period for the test to succeed. The default value is 0.05 (5%). +for the test to still pass. The default value is 0. Note: this key is parsed +but violation counting is not active in the current implementation; pass/fail +is determined solely by whether the peak GFLOPS meets the target. - - + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Config Key Type Description
target_stressFloat The maximum relative performance the GPU will attempt to achieve in gigaflops. This parameter is required.
This parameter indicates if each operation should copy the matrix data to the GPU before executing. The default value is true.
ramp_intervalIntegerThis is an time interval, specified in milliseconds, given to the test to +This is a time interval, specified in milliseconds, given to the test to reach the given target_stress gigaflops. The default value is 5000 (5 seconds). This time is counted against the duration of the test. If the target gflops, or stress, is not achieved in this time frame, the test will fail. If the target @@ -2092,250 +1721,271 @@ duration specified by the action, sustaining the stress load during that time.
toleranceFloat A value indicating how much the target_stress can fluctuate after the ramp -period for the test to succeed. The default value is 0.1 or 10%.
max_violationsInteger The number of tolerance violations that can occur after the ramp_interval -for the test to still pass. The default value is 0.
log_intervalInteger This is a positive integer, given in milliseconds, that specifies an -interval over which the moving average of the bandwidth will be calculated and +interval over which the moving average of GFLOPS will be calculated and logged.
matrix_sizeIntegerSize of the matrices of the SGEMM operations. The default value is -5760.
matrix_size_aIntegerNumber of rows of matrix A (the M dimension in GEMM). The default value is 5760.
matrix_size_bIntegerNumber of columns of matrix B (the N dimension in GEMM). The default value is 5760.
matrix_size_cIntegerInner (shared) dimension K of the GEMM operation (columns of A / rows of B). +The default value is 5760.
ops_typeStringGEMM operation type. Accepted values: sgemm, dgemm, +hgemm. If neither ops_type nor data_type is set, the +module defaults to sgemm. Mutually exclusive with data_type.
data_typeStringData type for the GEMM computation. Accepted values include: +fp4_r, fp6_r, fp8_r, bf16_r, fp16_r, +fp32_r, i8_r. If not specified +the module falls back to the type implied by ops_type.
out_data_typeStringOutput (result) data type, e.g. fp16_r, fp32_r. Only +applicable when using hipBLASLt. If not specified the default matches the +compute type.
compute_typeStringAccumulation/compute type used internally by the BLAS library. Accepted +values include fp32_r (default) and xf32_r (TF32 fast +compute).
blas_sourceStringBLAS library backend to use. Accepted values: +rocblas (default), hipblaslt.
hot_callsIntegerNumber of GEMM kernel invocations per measurement window used to amortise +launch overhead. The default value is 1.
matrix_initStringMatrix initialization method. Accepted values: +default – Initialize with default pattern (default). +trig – Initialize with trigonometric (sine/cosine) values. +rand – Initialize with random values.
transaIntegerTranspose operation applied to matrix A before the GEMM call. +0 = no transpose (default), 1 = transpose.
transbIntegerTranspose operation applied to matrix B before the GEMM call. +0 = no transpose, 1 = transpose (default).
alphaFloatScalar multiplier applied to the product of matrices A and B in the GEMM +operation (C = alpha * A * B + beta * C). The default value is 1.
betaFloatScalar multiplier applied to matrix C in the GEMM operation. The default +value is 1.
ldaIntegerLeading dimension offset added to the computed leading dimension of matrix +A. The default value is 0.
ldbIntegerLeading dimension offset added to the computed leading dimension of matrix +B. The default value is 0.
ldcIntegerLeading dimension offset added to the computed leading dimension of matrix +C. The default value is 0.
lddIntegerLeading dimension offset added to the computed leading dimension of matrix +D (output). The default value is 0.
scale_aStringScaling mode applied to matrix A. Used with low-precision types such as +fp8. Accepted values include block. If not specified no scaling is +applied.
scale_bStringScaling mode applied to matrix B. Used with low-precision types such as +fp8. Accepted values include block. If not specified no scaling is +applied.
rotatingIntegerSize of the rotating buffer (in elements) used to prevent data from +residing in cache between iterations, enabling cache-cold benchmarking. A +value of 0 disables rotating buffers. The default value is 0.
gemm_modeStringGEMM execution mode. Accepted values: +"" or unset – Standard (single) GEMM (default). +batched – Batched GEMM; use with batch_size. +strided_batched – Strided batched GEMM; use with batch_size and +stride_* keys.
batch_sizeIntegerNumber of GEMM operations in a batched or strided-batched call. Only +applicable when gemm_mode is batched or +strided_batched. The default value is 0.
stride_aIntegerStride (in elements) between consecutive matrices A in a strided-batched +GEMM. Only applicable when gemm_mode: strided_batched. The default +value is 0.
stride_bIntegerStride (in elements) between consecutive matrices B in a strided-batched +GEMM. The default value is 0.
stride_cIntegerStride (in elements) between consecutive matrices C in a strided-batched +GEMM. The default value is 0.
stride_dIntegerStride (in elements) between consecutive output matrices D in a +strided-batched GEMM. The default value is 0.
self_checkBoolIf true, validates the GEMM result for correctness after each operation. +Adds overhead; intended for debugging. The default value is false.
accuracy_checkBoolIf true, runs a numerical accuracy check after each GEMM operation. The +default value is false.
error_injectBoolIf true, enables error injection mode to deliberately introduce errors +into the computation for testing error detection. The default value is +false.
error_freqIntegerFrequency of error injection; specifies how often (every N operations) an +error is injected. Only applicable when error_inject: true. The +default value is 0.
error_countIntegerNumber of errors to inject per injection event. Only applicable when +error_inject: true. The default value is 0.
+ ### Output -Module specific output keys are described in the table below: - - - - - - - - - - - - - - - - - -
Output Key Type Description
target_stressTime Series FloatsThe average gflops over the last log interval.
max_gflopsFloatThe maximum sustained performance obtained by the GPU during the -test.
stress_violationsIntegerThe number of gflops readings that violated the tolerance of the test after -the ramp interval.
flops_per_opIntegerFlops (floating point operations) per operation queued to the GPU queue. -One operation is one call to SGEMM/DGEMM.
bytes_copied_per_opIntegerCalculated number of ops/second necessary to achieve target -gigaflops.
try_ops_per_secFloatCalculated number of ops/second necessary to achieve target -gigaflops.
passBool'true' if the GPU achieves its desired sustained performance -level.
- -An informational message indicating will be emitted when the test starts -execution: - - [INFO ][][] gst start copy matrix: - - -During the execution of the test, informational output providing the moving -average the GPU(s) gflops will be logged at each log_interval: - - [INFO ][][] gst Gflops: - -When the target gflops is achieved, the following message will be logged: - - [INFO ][][] gst target achieved - -If the target gflops, or stress, is not achieved in the “ramp_interval” -provided, the test will terminate and the following message will be logged: - - [INFO ][][] gst ramp time exceeded - -In this case the test will fail.\n +During the execution of the test, a result message reporting the GFLOPS +achieved by the GPU is logged at each `log_interval`: -If the target stress (gflops) is achieved the test will attempt to run for the -rest of the duration specified by the action, sustaining the stress load during -that time. If the stress level violates the bounds set by the tolerance level -during that time a violation message will be logged: + [] [GPU:: ] GFLOPS - [INFO ][][] gst stress violation +When the test completes, the final result message is printed: -When the test completes, the following result message will be printed: + [] [GPU:: ] GFLOPS Target GFLOPS: met: TRUE - [RESULT][][] gst Gflop: flops_per_op: bytes_copied_per_op: try_ops_per_sec: pass: - -The test will pass if the target_stress is reached before the end of the -ramp_interval and the stress_violations value is less than the given -max_violations value. Otherwise, the test will fail. +The test passes if `max_gflops >= target_stress * (1 - tolerance)` at the end +of the run. Otherwise the result is `met: FALSE`. ### Examples -When running the __GST__ module, users should provide at least an action name, -the module name (gst), a list of GPU IDs, the test duration and a target stress -value (gigaflops). Thus, the most basic configuration file looks like this: - - actions: - - name: action_gst_1 - module: gst - device: all - target_stress: 3500 - duration: 8000 - -For the above configuration file, all the missing configuration keys will have -their default -values (e.g.: __copy_matrix=true__, __matrix_size=5760__ etc.). For more -information about the default -values please consult the dedicated sections (__3.3 Common Configuration Keys__ -and __5.1 Configuration keys__). - -When the __RVS__ tool runs against such a configuration file, it will do the -following: - - run the stress test on all available (and compatible) AMD GPUs, one after -the other - - log a start message containing the GPU ID, the __target_stress__ and the -value of the __copy_matrix__:
- - [INFO ] [164337.932824] action_gst_1 gst 50599 start 3500.000000 copy matrix:true - - - emit, each __log_interval__ (e.g.: 1000ms), a message containing the -gigaflops value that the current GPU achieved:
- - [INFO ] [164355.111207] action_gst_1 gst 33367 Gflops 3535.670231 - - - log a message as soon as the current GPU reaches the given __target_stress__: - - [INFO ] [164350.804843] action_gst_1 gst 33367 target achieved 500.000000 - - - log a __ramp time exceeded__ message if the GPU was not able to reach the -__target_stress__ in the __ramp_interval__ time frame (e.g.: 5000). In such a -case, the test will also terminate:
- - [INFO ] [164013.788870] action_gst_1 gst 3254 ramp time exceeded 5000 - - - log the test result, when the stress test completes. The message contains -the test's overall result and some other statistics according to __5.2 Output -keys__:
- - [RESULT] [164355.647523] action_gst_1 gst 33367 Gflop: 4066.020766 flops_per_op: 382.205952x1e9 bytes_copied_per_op: 398131200 try_ops_per_sec: 9.157367 pass: TRUE +**Example:** - - log a __stress violation__ message when the current gigaflops (for the last -__log_interval__, e.g.; 1000ms) violates the bounds set by the __tolerance__ -configuration key (e.g.: 0.1). Please note that this message is not logged -during the __ramp_interval__ time frame:
+Run: - [INFO ] [164013.788870] action_gst_1 gst 3254 stress violation 2500 + ./rvs -c conf/MI355X/gst_single.conf -If a mandatory configuration key is missing, the __RVS__ tool will log an error -message and terminate the execution of the current module. For example, the -following configuration file will cause the __RVS__ to terminate with the -following error message:
__RVS-GST: action: action_gst_1 key -'target_stress' was not found__ +Configuration (`conf/MI355X/gst_single.conf`, first action): actions: - - name: action_gst_1 - module: gst + - name: gst-Tflops-2K2K2K-trig-fp4 device: all - duration: 8000 - -A more complex configuration file looks like this: - - actions: - - name: action_1 - device: 50599 33367 module: gst - parallel: false - count: 12 - wait: 100 - duration: 7000 - ramp_interval: 3000 - log_interval: 1000 - max_violations: 2 + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 copy_matrix: false - target_stress: 5000 - tolerance: 0.07 - matrix_size: 5760 - -For this configuration file, the RVS tool: - - will run the stress test only for the GPUs having the ID 50599 or 33367. To -get all the available GPU IDs, run __RVS__ tool with __-g__ option - - will run the test on the selected GPUs, one after the other - - will run each test, 12 times - - will only copy the matrices to the GPUs at the beginning of the test - - will wait 100ms before each test execution - - will try to reach 5000 gflops in maximum 3000ms - - if __target_stress__ (5000) is achieved in the __ramp_interval__ (3000 ms) -it will attempt to run the test for the rest of the duration, sustaining the -stress load during that time - - will allow a 7% __target_stress__ __tolerance__ (each __target_stress__ -violation will generate a __stress violation__ message as shown in the first -example) - - will allow only 2 __target_stress__ violations. Exceeding the -__max_violations__ will not terminate the test, but the __RVS__ will mark the -test result as "fail". - -The output for such a configuration key may look like this: - -__[INFO ] [172061.758830] action_1 gst 50599 start 5000.000000 copy -matrix:false__
-__[INFO ] [172063.547668] action_1 gst 50599 Gflops 6471.614725__
-__[INFO ] [172064.577715] action_1 gst 50599 target achieved 5000.000000__
-__[INFO ] [172065.609224] action_1 gst 50599 Gflops 5189.993529__
-__[INFO ] [172066.634360] action_1 gst 50599 Gflops 5220.373979__
-__[INFO ] [172067.659262] action_1 gst 50599 Gflops 5225.472000__
-__[INFO ] [172068.694305] action_1 gst 50599 Gflops 5169.935583__
-__[RESULT] [172069.573967] action_1 gst 50599 Gflop: 6471.614725 flops_per_op: -382.205952x1e9 bytes_copied_per_op: 398131200 try_ops_per_sec: 13.081952 pass: -TRUE__
-__[INFO ] [172069.574369] action_1 gst 33367 start 5000.000000 copy -matrix:false__
-__[INFO ] [172071.409483] action_1 gst 33367 Gflops 6558.348080__
-__[INFO ] [172072.438104] action_1 gst 33367 target achieved 5000.000000__
-__[INFO ] [172073.465033] action_1 gst 33367 Gflops 5215.285895__
-__[INFO ] [172074.501571] action_1 gst 33367 Gflops 5164.945297__
-__[INFO ] [172075.529468] action_1 gst 33367 Gflops 5210.207720__
-__[INFO ] [172076.558102] action_1 gst 33367 Gflops 5205.139424__
-__[RESULT] [172077.448182] action_1 gst 33367 Gflop: 6558.348080 flops_per_op: -382.205952x1e9 bytes_copied_per_op: 398131200 try_ops_per_sec: 13.081952 pass: -TRUE__
- -When setting the __parallel__ to false, the __RVS__ will run the stress tests on -all selected GPUs in parallel and the output may look like this: - -__[INFO ] [173381.407428] action_1 gst 50599 start 5000.000000 copy -matrix:false__
-__[INFO ] [173381.407744] action_1 gst 33367 start 5000.000000 copy -matrix:false__
-__[INFO ] [173383.245771] action_1 gst 33367 Gflops 6558.348080__
-__[INFO ] [173383.256935] action_1 gst 50599 Gflops 6484.532120__
-__[INFO ] [173384.274202] action_1 gst 33367 target achieved 5000.000000__
-__[INFO ] [173384.286014] action_1 gst 50599 target achieved 5000.000000__
-__[INFO ] [173385.301038] action_1 gst 33367 Gflops 5215.285895__
-__[INFO ] [173385.315794] action_1 gst 50599 Gflops 5200.080980__
-__[INFO ] [173386.337638] action_1 gst 33367 Gflops 5164.945297__
-__[INFO ] [173386.353274] action_1 gst 50599 Gflops 5159.964636__
-__[INFO ] [173387.365494] action_1 gst 33367 Gflops 5210.207720__
-__[INFO ] [173387.383437] action_1 gst 50599 Gflops 5195.032357__
-__[INFO ] [173388.401250] action_1 gst 33367 Gflops 5169.935583__
-__[INFO ] [173388.421599] action_1 gst 50599 Gflops 5154.993572__
-__[RESULT] [173389.282710] action_1 gst 33367 Gflop: 6558.348080 flops_per_op: -382.205952x1e9 bytes_copied_per_op: 398131200 try_ops_per_sec: 13.081952 pass: -TRUE__
-__[RESULT] [173389.305479] action_1 gst 50599 Gflop: 6484.532120 flops_per_op: -382.205952x1e9 bytes_copied_per_op: 398131200 try_ops_per_sec: 13.081952 pass: -TRUE__
- -It is important that all the configuration keys will be adjusted/fine-tuned -according to the actual GPUs and HW platform capabilities. For example, a matrix -size of 5760 should fit the VEGA 10 GPUs while 8640 should work with the VEGA 20 -GPUs. - -## IET Module + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp4_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + +Sample output (first action, abridged): + + [RESULT] [301995.91241 ] Action name :gst-Tflops-2K2K2K-trig-fp4 + [RESULT] [301995.175394] Module name :gst + [RESULT] [301995.794012] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 42583] Start of GPU ramp up + [RESULT] [302001.161732] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 42583] GFLOPS 536870 + [RESULT] [302002.161726] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 42583] End of GPU ramp up + [RESULT] [302005.176647] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 42583] GFLOPS 1059931 + [RESULT] [302008.187367] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 42583] GFLOPS 1055687 + [RESULT] [302011.194700] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 42583] GFLOPS 1056873 + [RESULT] [302014.199261] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 42583] GFLOPS 1063560 + [RESULT] [302017.167195] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 42583] GFLOPS 1063560 Target GFLOPS: 0 met: TRUE + [RESULT] [302017.168168] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 27226] Start of GPU ramp up + [RESULT] [302022.421856] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 27226] GFLOPS 536870 + [RESULT] [302023.421863] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 27226] End of GPU ramp up + [RESULT] [302026.422111] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 27226] GFLOPS 1065100 + [RESULT] [302029.437823] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 27226] GFLOPS 1065326 + [RESULT] [302032.447003] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 27226] GFLOPS 1067645 + [RESULT] [302035.454318] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 27226] GFLOPS 1068301 + [RESULT] [302038.430577] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 27226] GFLOPS 1068301 Target GFLOPS: 0 met: TRUE + [RESULT] [302038.431686] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 36479] Start of GPU ramp up + [RESULT] [302043.682076] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 36479] GFLOPS 554189 + ... + [RESULT] [302141.857878] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 11806] GFLOPS 1072748 + [RESULT] [302144.843455] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 11806] GFLOPS 1073132 Target GFLOPS: 0 met: TRUE + [RESULT] [302144.844774] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 57875] Start of GPU ramp up + [RESULT] [302150.131420] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 57875] GFLOPS 554189 + [RESULT] [302151.131405] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 57875] End of GPU ramp up + [RESULT] [302154.146042] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 57875] GFLOPS 1060026 + [RESULT] [302157.154100] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 57875] GFLOPS 1056631 + [RESULT] [302160.164765] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 57875] GFLOPS 1061420 + [RESULT] [302163.173056] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 57875] GFLOPS 1056547 + [RESULT] [302166.136604] [gst-Tflops-2K2K2K-trig-fp4] [GPU:: 57875] GFLOPS 1061420 Target GFLOPS: 0 met: TRUE + + +=====================================================================+ + | ROCm Validation Suite (RVS) Summary | + +=====================================================================+ + | System Overview | + +---------------------------------------------------------------------+ + | Operating System | Ubuntu 22.04.5 LTS | + | RVS version | 1.6.75 | + | ROCm version | 7.2.1-81 | + | amdgpu version | 6.16.13 | + | GPUs | 8 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 42583 | AMD Instinct MI355X - 27226 | + | 0 - 2 - 0000:05:00.0 | 1 - 3 - 0000:15:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 36479 | AMD Instinct MI355X - 17010 | + | 2 - 4 - 0000:65:00.0 | 3 - 5 - 0000:75:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 1590 | AMD Instinct MI355X - 51771 | + | 4 - 6 - 0000:85:00.0 | 5 - 7 - 0000:95:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 11806 | AMD Instinct MI355X - 57875 | + | 6 - 8 - 0000:e5:00.0 | 7 - 9 - 0000:f5:00.0 | + +=====================================================================+ + | Action Name | Module | Result | + +=====================================================================+ + | gst-Tflops-2K2K2K-trig-fp4 | GST | PASS | + +---------------------------------------------------------------------+ + +## IET module The Input EDPp Test can be used to characterize the peak power capabilities of a GPU (that is, TGP) for a sustained duration of time. This tool leverage GEMM workload @@ -2345,7 +1995,7 @@ that the GPUs can sustain a power level for a reasonable amount of time without like thermal violations arising. The test passes if GPU power meets or crosses the target power during the duration of the test else reported as fail. -This module should be used in conjunction with the GPU Monitor, to watch for +This module should be used in conjunction with GPU monitoring tools (for example, amd-smi), to watch for thermal, power and related anomalies while the target GPU(s) are under realistic load conditions. By setting the appropriate parameters a user can ensure that all GPUs in a node or cluster reach desired performance levels. Further analysis @@ -2353,205 +2003,743 @@ of the generated stats can also show variations in the required power, clocks or temperatures to reach these targets, and thus highlight GPUs or nodes that are operating less efficiently. -### Module Specific Keys +### Module specific keys Module specific keys are described in the table below: - - +
+
Config Key Type Description
+ - - - + - - -
Config Key Type Description
target_powerFloat This is a floating point value specifying the target sustained power level for the test.
ramp_intervalIntegerThis is an time interval, specified in milliseconds, given to the test to +This is a time interval, specified in milliseconds, given to the test to determine the compute load that will sustain the target power. The default value -is 5000 (5 seconds). This time is counted against the duration of the test. +is 5000 (5 seconds). Note: this key is parsed but not used by the worker in the +current implementation.
toleranceFloatA value indicating how much the target_power can fluctuate after the ramp -period for the test to succeed. The default value is 0.1 or 10%. +A value indicating how much the target_power can fluctuate for the test to +succeed. The default value is 0 (any violation immediately fails the test).
max_violationsIntegerThe number of tolerance violations that can occur after the ramp_interval -for the test to still pass. The default value is 0.
The number of tolerance violations that can occur for the test to still +pass. The default value is 0. Note: this key is parsed but not used by the +worker in the current implementation.
sample_intervalIntegerThe sampling rate for target_power values given in milliseconds. The default -value is 100 (.1 seconds). +The interval between power samples, specified in seconds. The default +value is 1 (1 second). If a value less than 1 is specified, it is raised to 1.
log_intervalIntegerThis is a positive integer, given in milliseconds, that specifies an -interval over which the moving average of the bandwidth will be calculated and -logged.
+This is a positive integer, given in milliseconds, that specifies a logging +interval. Note: this key is parsed but not used by the worker in the current +implementation. +cp_workloadBool +If true, enables the GEMM compute workload to drive GPU power. This is the +primary workload used to reach the target power level. The default value is +true. -### Output +bw_workloadBool +If true, enables a memory bandwidth kernel workload in addition to or +instead of the GEMM workload. The default value is false. -Module specific output keys are described in the table below: +wg_countInteger +Number of GPU workgroups used for the bandwidth kernel. Only applicable +when bw_workload: true. The default value is 80. - - - - - - - - + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Output Key Type Description
current_powerTime Series FloatsThe current measured power of the GPU.
power_violationsIntegerThe number of power reading that violated the tolerance of the test after -the ramp interval. -
passBool'true' if the GPU achieves its desired sustained power level in the ramp -interval.
nt_loadsBoolIf true, the bandwidth kernel uses non-temporal loads that bypass the L2 +cache, exercising memory bandwidth more directly. Only applicable when +bw_workload: true. The default value is false.
matrix_sizeIntegerSets all three GEMM matrix dimensions (M, N, and K) to the same value. +Equivalent to setting matrix_size_a, matrix_size_b, and +matrix_size_c to the same value. The default value is 5760.
matrix_size_aIntegerNumber of rows of matrix A (the M dimension in GEMM). Overrides +matrix_size for this dimension. Default is 0 (uses +matrix_size).
matrix_size_bIntegerNumber of columns of matrix B (the N dimension in GEMM). Overrides +matrix_size for this dimension. Default is 0 (uses +matrix_size).
matrix_size_cIntegerInner (shared) dimension K of the GEMM operation. Overrides +matrix_size for this dimension. Default is 0 (uses +matrix_size).
ops_typeStringGEMM operation type. Accepted values: sgemm, dgemm, +hgemm. If neither ops_type nor data_type is set, the +module defaults to sgemm. Mutually exclusive with +data_type.
data_typeStringData type for the GEMM computation. Accepted values include: +fp4_r, fp6_r, fp8_r, bf16_r, fp16_r, +fp32_r, i8_r. If not specified +the module falls back to the type implied by ops_type.
out_data_typeStringOutput (result) data type, e.g. fp16_r, fp32_r. Only +applicable when using hipBLASLt. If not specified the default matches the +compute type.
compute_typeStringAccumulation/compute type used internally by the BLAS library. Accepted +values include fp32_r (default) and xf32_r (TF32 fast +compute).
blas_sourceStringBLAS library backend to use. Accepted values: +rocblas (default), hipblaslt.
hot_callsIntegerNumber of GEMM kernel invocations per measurement window used to amortise +launch overhead. The default value is 1.
matrix_initStringMatrix initialization method. Accepted values: +default – Initialize with default pattern (default). +trig – Initialize with trigonometric (sine/cosine) values. +rand – Initialize with random values.
transaIntegerTranspose operation applied to matrix A before the GEMM call. +0 = no transpose (default), 1 = transpose.
transbIntegerTranspose operation applied to matrix B before the GEMM call. +0 = no transpose, 1 = transpose (default).
alphaFloatScalar multiplier applied to the product of matrices A and B +(C = alpha * A * B + beta * C). The default value is 1.
betaFloatScalar multiplier applied to matrix C in the GEMM operation. The default +value is 1.
ldaIntegerLeading dimension offset for matrix A. The default value is 0.
ldbIntegerLeading dimension offset for matrix B. The default value is 0.
ldcIntegerLeading dimension offset for matrix C. The default value is 0.
lddIntegerLeading dimension offset for matrix D (output). The default value is +0.
gemm_modeStringGEMM execution mode. Accepted values: +"" or unset – Standard (single) GEMM (default). +batched – Batched GEMM; use with batch_size. +strided_batched – Strided batched GEMM; use with batch_size +and stride_* keys.
batch_sizeIntegerNumber of GEMM operations in a batched or strided-batched call. Only +applicable when gemm_mode is batched or +strided_batched. The default value is 0.
stride_aIntegerStride (in elements) between consecutive matrices A in a +strided-batched GEMM. The default value is 0.
stride_bIntegerStride (in elements) between consecutive matrices B in a +strided-batched GEMM. The default value is 0.
stride_cIntegerStride (in elements) between consecutive matrices C in a +strided-batched GEMM. The default value is 0.
stride_dIntegerStride (in elements) between consecutive output matrices D in a +strided-batched GEMM. The default value is 0.
+ -### Examples -**Example 1:** +### Output -A regular IET configuration file looks like this: +During the execution of the test, a result message reporting the measured GPU +power is logged at each `sample_interval`: - actions: - - name: action_1 - device: all - module: iet - parallel: false - count: 2 - wait: 100 - duration: 10000 - ramp_interval: 5000 - sample_interval: 500 - log_interval: 500 - max_violations: 1 - target_power: 135 - tolerance: 0.1 - matrix_size: 5760 + [] [GPU:: ] Power(W) + +When the test completes, the pass/fail result is printed: -*Please note:* -- when setting the 'device' configuration key to 'all', the RVS will detect all the AMD compatible GPUs and run the test on all of them -- the test will run 2 times on each GPU (count = 2) -- only one power violation is allowed. If the total number of violations is bigger than 1 the IET test result will be marked as 'failed' + [] [GPU:: ] pass: TRUE -When the RVS tool runs against such a configuration file, it will do the following: -- run the test on all AMD compatible GPUs +The test passes if the peak power measured during the run meets or exceeds +`target_power * (1 - tolerance)`. When JSON output is enabled (`-j`), an +`average power` field is also included in the result record. -- log a start message containing the GPU ID and the target_power, e.g.: +### Examples - [INFO ] [167316.308057] action_1 iet 50599 start 135.000000 +**Example:** -- emit, each log_interval (e.g.: 500ms), a message containing the power for the current GPU +Run: - [INFO ] [167319.266707] action_1 iet 50599 current power 136.878342 + ./rvs -c conf/MI355X/iet_stress.conf -- log a message as soon as the current GPU reaches the given target_power +Configuration (`conf/MI355X/iet_stress.conf`, first action): - [INFO ] [167318.793062] action_1 iet 50599 target achieved 135.000000 + actions: + - name: iet-stress-1400W-true + device: all + module: iet + parallel: true + duration: 600000 + ramp_interval: 1000 + sample_interval: 5000 + log_interval: 5000 + target_power: 1400 + tolerance: 0.01 + bw_workload: true + cp_workload: false + wg_count: 256 + nt_loads: true + +Sample output (first action, abridged): + + [RESULT] [302166.984877] Action name :iet-stress-1400W-true + [RESULT] [302167.47707 ] Module name :iet + [RESULT] [302167.678963] [iet-stress-1400W-true] [GPU:: 42583] Power(W) 241.000000 + [RESULT] [302167.679266] [iet-stress-1400W-true] [GPU:: 36479] Power(W) 238.000000 + [RESULT] [302167.679363] [iet-stress-1400W-true] [GPU:: 17010] Power(W) 242.000000 + [RESULT] [302167.679366] [iet-stress-1400W-true] [GPU:: 27226] Power(W) 251.000000 + [RESULT] [302167.686277] [iet-stress-1400W-true] [GPU:: 1590] Power(W) 244.000000 + [RESULT] [302167.686419] [iet-stress-1400W-true] [GPU:: 57875] Power(W) 247.000000 + [RESULT] [302167.686591] [iet-stress-1400W-true] [GPU:: 11806] Power(W) 243.000000 + [RESULT] [302167.686605] [iet-stress-1400W-true] [GPU:: 51771] Power(W) 239.000000 + [RESULT] [302172.680300] [iet-stress-1400W-true] [GPU:: 36479] Power(W) 1398.000000 + [RESULT] [302172.680303] [iet-stress-1400W-true] [GPU:: 42583] Power(W) 1398.000000 + [RESULT] [302172.680570] [iet-stress-1400W-true] [GPU:: 27226] Power(W) 1398.000000 + [RESULT] [302172.680606] [iet-stress-1400W-true] [GPU:: 17010] Power(W) 1398.000000 + [RESULT] [302172.687273] [iet-stress-1400W-true] [GPU:: 57875] Power(W) 1399.000000 + [RESULT] [302172.687390] [iet-stress-1400W-true] [GPU:: 1590] Power(W) 1399.000000 + [RESULT] [302172.687638] [iet-stress-1400W-true] [GPU:: 11806] Power(W) 1400.000000 + [RESULT] [302172.687845] [iet-stress-1400W-true] [GPU:: 51771] Power(W) 1399.000000 + [RESULT] [302177.681560] [iet-stress-1400W-true] [GPU:: 36479] Power(W) 1395.000000 + [RESULT] [302177.681561] [iet-stress-1400W-true] [GPU:: 42583] Power(W) 1393.000000 + [RESULT] [302177.681698] [iet-stress-1400W-true] [GPU:: 27226] Power(W) 1394.000000 + [RESULT] [302177.681724] [iet-stress-1400W-true] [GPU:: 17010] Power(W) 1395.000000 + ... + [RESULT] [302347.724033] [iet-stress-1400W-true] [GPU:: 42583] Power(W) 1399.000000 + [RESULT] [302347.724905] [iet-stress-1400W-true] [GPU:: 27226] pass: TRUE + [RESULT] [302347.727817] [iet-stress-1400W-true] [GPU:: 17010] pass: TRUE + [RESULT] [302347.729796] [iet-stress-1400W-true] [GPU:: 42583] pass: TRUE + [RESULT] [302347.730984] [iet-stress-1400W-true] [GPU:: 11806] Power(W) 1400.000000 + [RESULT] [302347.730984] [iet-stress-1400W-true] [GPU:: 1590] Power(W) 1401.000000 + [RESULT] [302347.730997] [iet-stress-1400W-true] [GPU:: 57875] Power(W) 1400.000000 + [RESULT] [302347.731021] [iet-stress-1400W-true] [GPU:: 51771] Power(W) 1399.000000 + [RESULT] [302347.732052] [iet-stress-1400W-true] [GPU:: 36479] pass: TRUE + [RESULT] [302347.737108] [iet-stress-1400W-true] [GPU:: 1590] pass: TRUE + [RESULT] [302347.739075] [iet-stress-1400W-true] [GPU:: 11806] pass: TRUE + [RESULT] [302347.740868] [iet-stress-1400W-true] [GPU:: 51771] pass: TRUE + [RESULT] [302347.742768] [iet-stress-1400W-true] [GPU:: 57875] pass: TRUE + + +=====================================================================+ + | ROCm Validation Suite (RVS) Summary | + +=====================================================================+ + | System Overview | + +---------------------------------------------------------------------+ + | Operating System | Ubuntu 22.04.5 LTS | + | RVS version | 1.6.75 | + | ROCm version | 7.2.1-81 | + | amdgpu version | 6.16.13 | + | GPUs | 8 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 42583 | AMD Instinct MI355X - 27226 | + | 0 - 2 - 0000:05:00.0 | 1 - 3 - 0000:15:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 36479 | AMD Instinct MI355X - 17010 | + | 2 - 4 - 0000:65:00.0 | 3 - 5 - 0000:75:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 1590 | AMD Instinct MI355X - 51771 | + | 4 - 6 - 0000:85:00.0 | 5 - 7 - 0000:95:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 11806 | AMD Instinct MI355X - 57875 | + | 6 - 8 - 0000:e5:00.0 | 7 - 9 - 0000:f5:00.0 | + +=====================================================================+ + | Action Name | Module | Result | + +=====================================================================+ + | iet-stress-1400W-true | IET | PASS | + +---------------------------------------------------------------------+ + + +## Pulse Module + +```{warning} +This module is in beta and is not intended for production use. Pass/fail criteria (especially around power-delta enforcement, clock-pinning verification, and throttle detection) are still being tuned and may change between releases. +``` + +The **Pulse** module is intended for **time-varying** GPU power stress: it alternates **high** phases (maximum clocks plus continuous GEMM) and **low** phases (minimum clocks plus idle/sleep), repeating for the action **duration**. That produces periodic power swings useful for exercising PSU transient response and platform power delivery, complementing **IET** (which targets a sustained power level). + +GEMM type, matrix size, and BLAS backend follow the same concepts as **GST** / **IET** (see those sections and `rvs_blas`). Sample reference configuration: `rvs/conf/pulse_single.conf`. + +### Behavior summary + +- Each **pulse cycle** is one high phase followed by one low phase. Phase lengths derive from `pulse_rate` (Hz) and `high_phase_ratio`; each phase is at least 10 ms after rounding. +- **High phase:** sets GPU clocks high, runs GEMM in a loop (`workload_iterations` per inner batch) until the high-phase time budget elapses, samples power, checks junction temperature (**failure if above 105°C**). +- After the high phase, the module calls `hipDeviceSynchronize` so work drains before the low phase (clearer separation of high vs. low power). +- **Low phase:** sets clocks low, sleeps briefly between `sample_interval`-style sampling (5 ms sleep steps in the implementation), samples power. +- With `parallel: true` and more than one GPU, threads coordinate with a `std::barrier` and a small GPU kernel using fine-grained coherent host memory so all GPUs synchronize across the pulse loop (avoids deadlock on shutdown via a shared done flag). +- **Pass:** for primary MCM dies, the action passes if at least one pulse completed, there was **no** BLAS enqueue/sync failure (unless `halt_on_error` stops earlier), and no thermal violation; secondary MCM GPUs are treated as pass by default. Fail otherwise. + +### Prerequisites + +- ROCm with hipBLASLt and rocBLAS available as for the rest of RVS (the `rvs` binary links hipblaslt; GEMM code lives in rvslib). +- AMD SMI initialized for power and temperature queries (same family of requirements as other SMI-based modules). Elevated privileges are often required for clock control and power metrics. +- For hipBLASLt, `data_type` must be valid (for example, `fp32_r` with `sgemm`, `fp16_r` with `hgemm` and `compute_type` such as `fp32_r`). If `data_type` is omitted with `blas_source: hipblaslt`, the module infers `fp32_r`/`fp64_r`/`fp16_r` from `ops_type`/`sgemm`/`dgemm`/`hgemm` respectively; for `dgemm` with hipBLASLt, if `compute_type` is still the default `fp32_r`, it is adjusted to `fp64_r`. + +### Module specific keys + +Keys below are in addition to common keys (`name`, `module`, `device`, `duration`, `parallel`, `log_interval`, `count`, `wait` and so on. For more information, see Common configuration keys. + +
+ + + + + + + + + + + + + + + + + + + + + + + + + + + +
Config Key TypeDescription
pulse_rateIntegerPulse frequency in **Hertz** (cycles per second). Default 2. Use roughly **1–5 Hz** if you want power readings to show clear high/low separation; very high rates shorten phases below typical GPU power-state and SMI averaging windows, so reported averages may converge.
high_phase_ratioFloatFraction of each cycle spent in the **high** (GEMM) phase, **0.0–1.0**. Default 0.5. Lower values spend more time in the low phase.
matrix_sizeIntegerSquare GEMM dimension **M = N = K**. Default 4096. Larger matrices increase compute and power draw during the high phase.
ops_typeStringGEMM operation flavor (e.g. sgemm, dgemm, hgemm). Default sgemm. Must match **data_type** / **compute_type** when using hipBLASLt.
data_typeStringMatrix data type string for hipBLASLt (e.g. fp32_r, fp16_r). Default empty (rocBLAS can infer from **ops_type** alone).
out_data_typeStringOptional output type for GEMM; default empty (same as input type where applicable).
compute_typeStringhipBLASLt compute type (e.g. fp32_r). Default fp32_r.
blas_sourceStringrocblas or hipblaslt. Default rocblas.
alphaFloatGEMM scalar α. Default 2.0.
betaFloatGEMM scalar β. Default -1.0.
transaIntegerTranspose A: **0** no transpose, **1** transpose. Default 0.
transbIntegerTranspose B: **0** no transpose, **1** transpose. Default 1.
ldaIntegerLeading dimension offset A (0 = use minimum). Defaults 0.
ldbIntegerLeading dimension offset B. Default 0.
ldcIntegerLeading dimension offset C. Default 0.
lddIntegerLeading dimension offset D. Default 0.
matrix_initStringHost matrix initialization (e.g. default, hiprand). Default default.
workload_iterationsIntegerGEMM calls per inner batch in the high phase before re-checking time and power. Default 128. Increase for heavier per-iteration work; tune with **pulse_rate** and phase length.
sample_intervalIntegerReserved sampling interval (ms) for pulse-specific logic; values below **50** are raised to **50**. Default 100.
toleranceFloatDefault 10.0. Parsed and passed to the worker; **not** currently used in pass/fail logic (reserved for future checks).
verify_modeStringe.g. diff / crc. Default diff. Parsed; **not** currently used in pass/fail logic (reserved for future GEMM verification).
halt_on_errorBoolIf true, stop the GPU thread on first BLAS or thermal error. Default false.
hot_callsIntegerBLAS “hot call” / warmup-related parameter forwarded to **rvs_blas**. Default 1.
gpu_sync_waitIntegerDefault 10000. Parsed from configuration; **not** referenced by the current barrier implementation (placeholder for future timeout behavior).
max_temp_cFloatJunction temperature ceiling in degrees Celsius. If the GPU junction temperature exceeds this threshold during the run, the worker logs a thermal-violation error and, when halt_on_error is true, terminates that GPU thread. Default 105.0. A value of 0 disables thermal checking entirely.
+
-- log a 'ramp time exceeded' message if the GPU was not able to reach the target_power in the ramp_interval time frame (e.g.: 5000ms). In such a case, the test will also terminate +### Output - [INFO ] [167648.832413] action_1 iet 50599 ramp time exceeded 5000 +Log lines use the action name, module tag pulse, and GPU ID. + +
+ + + + + + + + +
Output / log Description
Start[INFO] ... pulse <gpu_id> start pulse_rate=<hz>
Parameters[INFO] ... pulse_rate=... Hz period=...ms high=...ms low=...ms
Thermal error[ERROR] ... thermal violation: <temp>C (limit <max_temp_c>C) (emitted only when max_temp_c is > 0)
Periodic summaryAt each log_interval, moving averages and extrema: pulse #N avg_high=...W avg_low=...W max_high=...W min_low=...W delta=...W
CompletionSummary over the run: pulse count, average high/low power, delta, max high, min low.
pass[RESULT] ... pass: true or false per GPU.
+
-- log a 'power violation message' when the current power (for the last sample_interval, e.g.; 500ms) violates the bounds set by the tolerance configuration key (e.g.: 0.1). Please note that this message is never logged during the ramp_interval time frame +With JSON logging enabled, per-pulse records can include `power_high_w`, `power_low_w`, `power_delta_w`, phase durations, `temp_c`, `gemm_count`, and a final summary with pass. - [INFO ] [161251.971277] action_1 iet 3254 power violation 73.783211 +### Example -- log the test result, when the stress test completes. +Minimal illustration (see `pulse_single.conf` for full examples including hipblaslt + `fp16_r`): - [RESULT] [167305.260051] action_1 iet 33367 pass: TRUE + actions: + - name: pulse_stress_basic + device: all + module: pulse + parallel: true + duration: 60000 + log_interval: 5000 + pulse_rate: 2 + high_phase_ratio: 0.5 + ops_type: sgemm + matrix_size: 8192 + blas_source: hipblaslt + data_type: fp32_r + compute_type: fp32_r + workload_iterations: 128 + halt_on_error: false + +Run from the build or package `bin` directory, for example: + + ./rvs -c conf/pulse_single.conf -d 3 + + +```{note} +- Tune `matrix_size`, `pulse_rate`, and `high_phase_ratio` to match GPU class and the transient behavior you want to stress. +- hipBLASLt often delivers higher GEMM throughput than rocBLAS on supported GPUs; `fp16_r`/`hgemm` with `compute_type`: `fp32_r` is a common high-throughput choice (as in the `pulse_stress_fp16` action in `pulse_single.conf`). +``` + +## MEM module + +The Memory module tests GPU memory for hardware errors and soft errors using +HIP. It executes a configurable suite of memory test algorithms that exercise +various data patterns and access sequences. Each test can be individually +included or excluded. The module reports errors found per test and passes only +if no memory errors are detected. + +The following tests are available (referenced by index in `exclude`): + +| Index | Test Name | +|---|---| +| 0 | Walking 1 bit | +| 1 | Own address test | +| 2 | Moving inversions, ones & zeros | +| 3 | Moving inversions, 8-bit pattern | +| 4 | Moving inversions, random pattern | +| 5 | Block move, 64 moves | +| 6 | Moving inversions, 32-bit pattern | +| 7 | Random number sequence | +| 8 | Modulo 20, random pattern | +| 9 | Bit fade test | +| 10 | Memory stress test | + +### Module specific keys + +
+ + + + + + + + + + + + + + + + +
Config Key Type Description
mem_blocksIntegerNumber of GPU memory blocks used per test iteration. The default value is +256.
num_passesIntegerNumber of passes (repeats) per block in each test. The default value is +1.
thrds_per_blkIntegerNumber of HIP threads per block launched for each test kernel. The default +value is 128.
stressBoolIf true, intended to enable the memory stress test (Test 10) in addition to +the standard tests. The default value is false. Note: this key is parsed but +does not affect which tests run in the current implementation; all tests are +always executed.
mapped_memoryBoolIf true, uses host-mapped (pinned) memory instead of device memory for the +test buffers. The default value is false.
num_iterIntegerNumber of iterations to run per test. The default value is 1.
excludeCollection of IntegersSpace-separated list of test indices (0–10) to skip. For example, +exclude: 9 10 is intended to skip the bit fade and memory stress tests. +Note: this key is parsed but does not affect which tests run in the current +implementation; all tests are always executed.
+
-The output for such a configuration file may look like this: +### Output - [INFO ] [167261.27161 ] action_1 iet 33367 start 135.000000 - [INFO ] [167263.516803] action_1 iet 33367 current power 136.934479 - [INFO ] [167263.521355] action_1 iet 33367 target achieved 135.000000 - [INFO ] [167264.16925 ] action_1 iet 33367 current power 138.421844 - [INFO ] [167264.517018] action_1 iet 33367 current power 138.394608 - ... - [INFO ] [167271.518402] action_1 iet 33367 current power 139.231918 - [RESULT] [167272.67686 ] action_1 iet 33367 pass: TRUE - [INFO ] [167272.68029 ] action_1 iet 3254 start 135.000000 - [INFO ] [167274.552026] action_1 iet 3254 current power 139.363525 - [INFO ] [167274.552059] action_1 iet 3254 target achieved 135.000000 - [INFO ] [167275.52168 ] action_1 iet 3254 current power 138.661453 - [INFO ] [167275.552241] action_1 iet 3254 current power 138.857635 - ... - [INFO ] [167282.553983] action_1 iet 3254 current power 140.069687 - [RESULT] [167283.95763 ] action_1 iet 3254 pass: TRUE - [INFO ] [167283.96158 ] action_1 iet 50599 start 135.000000 - [INFO ] [167285.532999] action_1 iet 50599 current power 137.205032 - [INFO ] [167285.543084] action_1 iet 50599 target achieved 135.000000 - [INFO ] [167286.33050 ] action_1 iet 50599 current power 136.137115 - ... - [INFO ] [167293.534672] action_1 iet 50599 current power 139.753464 - [RESULT] [167294.131420] action_1 iet 50599 pass: TRUE +
+ + + + + + + + + + +
Output Key Type Description
TestStringName of the memory test that was executed.
Time TakenFloatTime in seconds taken to complete the test on this GPU.
errorsIntegerNumber of memory errors detected during the test. A value of 0 indicates +no errors.
passBoolTrue if no memory errors were detected for this test.
+
+ +### Examples +**Example:** -**Example 2:** +Run: -Another configuration file, which may raise some 'power violation' messages (due to the small tolerance value) looks like this + ./rvs -c conf/mem.conf +Configuration (`conf/mem.conf`, first action): + + actions: - name: action_1 device: all - module: iet - parallel: false + module: mem + parallel: true count: 1 wait: 100 - duration: 8000 - ramp_interval: 5000 - sample_interval: 700 - log_interval: 700 - max_violations: 1 - target_power: 80 - tolerance: 0.06 - matrix_size: 5760 - -The output for such a configuration file may look like this: - - [INFO ] [161236.677785] action_1 iet 33367 start 80.000000 - [INFO ] [161239.350055] action_1 iet 33367 current power 84.186142 - [INFO ] [161239.354542] action_1 iet 33367 target achieved 80.000000 - ... - [INFO ] [161241.450517] action_1 iet 33367 current power 77.001945 - [INFO ] [161241.459600] action_1 iet 33367 power violation 75.163689 - [INFO ] [161242.150642] action_1 iet 33367 current power 82.063576 - [RESULT] [161245.698113] action_1 iet 33367 pass: TRUE - [INFO ] [161245.698525] action_1 iet 3254 start 80.000000 - [INFO ] [161248.394003] action_1 iet 3254 current power 78.842796 - [INFO ] [161248.418631] action_1 iet 3254 target achieved 80.000000 - [INFO ] [161249.94149 ] action_1 iet 3254 current power 79.938454 - ... - [INFO ] [161249.794201] action_1 iet 3254 current power 76.511711 - [INFO ] [161249.818803] action_1 iet 3254 power violation 74.279594 - [INFO ] [161250.494263] action_1 iet 3254 current power 74.615120 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 50000 + +Sample output (first action, abridged): + + [RESULT] [302697.357999] Action name :action_1 + [RESULT] [302697.415572] Module name :mem + [RESULT] [302697.637264] [action_1] mem The following memory tests will run + [RESULT] [302697.637269] =============== Test 1 [Walking 1 bit] + [RESULT] [302697.637270] =============== Test 2 [Own address test] + [RESULT] [302697.637270] =============== Test 3 [Moving inversions, ones&zeros] + [RESULT] [302697.637271] =============== Test 4 [Moving inversions, 8 bit pat] + [RESULT] [302697.637271] =============== Test 5 [Moving inversions, random pattern] + [RESULT] [302697.637271] =============== Test 6 [Block move, 64 moves] + [RESULT] [302697.637271] =============== Test 7 [Moving inversions, 32 bit pat] + [RESULT] [302697.637272] =============== Test 8 [Random number sequence] + [RESULT] [302697.637272] =============== Test 9 [Modulo 20, random pattern] + [RESULT] [302697.869004] [action_1] mem Test 1: Change one bit memory addresss + [RESULT] [302697.872232] [action_1] mem Test 1: Change one bit memory addresss + [RESULT] [302697.890031] [action_1] mem Test 1: Change one bit memory addresss + [RESULT] [302697.893155] [action_1] mem Test 1: Change one bit memory addresss + [RESULT] [302697.895754] [action_1] mem Test 1 : PASS + [RESULT] [302697.895771] [action_1] mem Test 2: Each Memory location is filled with its own address + [RESULT] [302697.896227] [action_1] mem Test 1: Change one bit memory addresss + [RESULT] [302697.900155] [action_1] mem Test 1 : PASS + [RESULT] [302697.900173] [action_1] mem Test 2: Each Memory location is filled with its own address + [RESULT] [302697.903915] [action_1] mem Test 1: Change one bit memory addresss ... - [INFO ] [161254.117386] action_1 iet 3254 power violation 73.682312 - [RESULT] [161254.738939] action_1 iet 3254 pass: FALSE - [INFO ] [161254.739387] action_1 iet 50599 start 80.000000 - [INFO ] [161257.374079] action_1 iet 50599 current power 81.560165 - [INFO ] [161257.392085] action_1 iet 50599 target achieved 80.000000 - [INFO ] [161258.774304] action_1 iet 50599 current power 75.057304 - ... - [INFO ] [161262.974833] action_1 iet 50599 current power 80.200668 - [RESULT] [161263.771631] action_1 iet 50599 pass: TRUE - - -*Important notes:* - - -- all the missing configuration keys (if any) will have their default values. For more information about the default values please consult the dedicated sections (3.3 Common Configuration Keys and 13.1 Module specific keys). - - -- if a mandatory configuration key is missing, the RVS tool will log an error message and terminate the execution of the current module. For example, if the target_power is missing, the RVS to terminate with the following error message: "RVS-IET: action: action_1 key 'target_power' was not found" - + [RESULT] [302732.976817] [action_1] mem Test 11: elapsedtime = 1986.538940 bandwidth = 6443.431641GB/s + [RESULT] [302732.978980] [action_1] mem Test 11 : PASS + [RESULT] [302733.171129] [action_1] mem Test 11: elapsedtime = 1989.459717 bandwidth = 6433.972168GB/s + [RESULT] [302733.173334] [action_1] mem Test 11 : PASS + [RESULT] [302733.208421] [action_1] mem Test 11: elapsedtime = 1974.268921 bandwidth = 6483.477539GB/s + [RESULT] [302733.210793] [action_1] mem Test 11 : PASS + [RESULT] [302730.722708] [action_1] mem Test 11: elapsedtime = 1964.402588 bandwidth = 6516.041016GB/s + [RESULT] [302730.724818] [action_1] mem Test 11 : PASS + [RESULT] [302731.757998] [action_1] mem Test 11: elapsedtime = 2002.477539 bandwidth = 6392.145508GB/s + [RESULT] [302731.760200] [action_1] mem Test 11 : PASS + + +=====================================================================+ + | ROCm Validation Suite (RVS) Summary | + +=====================================================================+ + | System Overview | + +---------------------------------------------------------------------+ + | Operating System | Ubuntu 22.04.5 LTS | + | RVS version | 1.6.75 | + | ROCm version | 7.2.1-81 | + | amdgpu version | 6.16.13 | + | GPUs | 8 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 42583 | AMD Instinct MI355X - 27226 | + | 0 - 2 - 0000:05:00.0 | 1 - 3 - 0000:15:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 36479 | AMD Instinct MI355X - 17010 | + | 2 - 4 - 0000:65:00.0 | 3 - 5 - 0000:75:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 1590 | AMD Instinct MI355X - 51771 | + | 4 - 6 - 0000:85:00.0 | 5 - 7 - 0000:95:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 11806 | AMD Instinct MI355X - 57875 | + | 6 - 8 - 0000:e5:00.0 | 7 - 9 - 0000:f5:00.0 | + +=====================================================================+ + | Action Name | Module | Result | + +=====================================================================+ + | action_1 | MEM | PASS | + +---------------------------------------------------------------------+ + + +## BABEL module + +The BABEL module executes BabelStream benchmark tests that measure GPU memory +bandwidth. BabelStream is a synthetic benchmark based on the STREAM benchmark +for CPUs. It runs configurable memory kernels (Copy, Mul, Add, Triad, Dot, +Read, Write) using HIP and reports the achieved bandwidth in GB/s or GiB/s for +each kernel. The benchmark is useful for characterizing peak memory bandwidth +and detecting bandwidth regressions. + +### Module specific keys + +
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Config Key Type Description
array_sizeIntegerNumber of elements in the test array (element count, not bytes). The actual +memory size depends on the data type selected by test_type. The default +value is 33554432 (32 M elements).
test_typeIntegerData precision used for the benchmark. Accepted values: +1 – Float (32-bit, default). +2 – Double (64-bit). +3 – Triad float. +4 – Triad double.
num_iterIntegerNumber of kernel launch iterations. Used when duration is 0. The +default value is 100.
durationIntegerDuration of the test in milliseconds. When greater than 0, the test runs +for this time instead of a fixed number of iterations. A value of 0 means use +num_iter. The default value is 0.
mibibytesBoolIf true, bandwidth is reported in GiB/s. If false, GB/s are used. This +key affects the reporting unit only; it does not change how array_size +is interpreted. The default value is false.
o/p_csvBoolIf true, outputs results in CSV format in addition to the standard log. +The default value is false.
readBoolIf true, enables the Read kernel (measures read-only bandwidth). The +default value is false.
writeBoolIf true, enables the Write kernel (measures write-only bandwidth). The +default value is false.
copyBoolIf true, enables the Copy kernel (a[i] = b[i]). The default value is +false.
mulBoolIf true, enables the Mul kernel (a[i] = scalar * b[i]). The default value +is false.
addBoolIf true, enables the Add kernel (a[i] = b[i] + c[i]). The default value +is false.
dotBoolIf true, enables the Dot kernel (sum of a[i] * b[i]). The default value +is false.
triadBoolIf true, enables the Triad kernel (a[i] = b[i] + scalar * c[i]). The +default value is false.
data_initStringData initialization method for the test arrays. Accepted values: +default – Initialize with default constant values (default). +gpu_norm_dist – Initialize on GPU with a normal distribution. +cpu_norm_dist – Initialize on CPU with a normal distribution. +zero_init – Initialize all elements to zero.
nontemporalStringNon-temporal (streaming) memory access mode for the kernels. Non-temporal +stores bypass the cache and can be useful for measuring true memory bandwidth. +Accepted values: +all – Apply non-temporal accesses to all kernels (default). +none – Disable non-temporal accesses. +read – Apply only to read accesses. +write – Apply only to write accesses.
dwords_per_laneIntegerNumber of 32-bit words processed per GPU lane per kernel invocation. The +default value is 4.
chunks_per_blockIntegerNumber of data chunks processed per GPU thread block. The default value +is 2.
tb_sizeIntegerThread block size (number of threads per block). The default value is +1024.
sustainedBoolIf true, enables sustained mode: all kernels are launched back-to-back without +per-iteration synchronization, and a single synchronization is issued +after all iterations complete. Bandwidth is reported as the average across all +iterations in all four output columns (MBytes/sec, Max, Min, Avg). This mode +measures steady-state memory bandwidth under continuous load. The default value +is false.
+
-- it is important that all the configuration keys will be adjusted/fine-tuned according to the actual GPUs and HW platform capabilities. +### Output +
+ + + + + + + + + + +
Output Key Type Description
FunctionStringName of the BabelStream kernel executed (e.g., Copy, Mul, Add, Triad, +Dot, Read, Write).
MBytes/secFloatMeasured memory bandwidth for this kernel in MB/s (or MiB/s if +mibibytes: true).
Max_MBytes/secFloatMaximum memory bandwidth observed across all iterations for this +kernel.
passBoolTrue if the kernel completed successfully.
+
-*) for example, a matrix size of 5760 should fit the VEGA 10 GPUs while 8640 should work with the VEGA 20 GPUs +### Examples +**Example:** -*) for small target_power values (e.g.: 30-40W), the sample_interval should be increased, otherwise the IET may fail either to achieve the given target_power or to sustain it (e.g.: ramp_interval = 1500 for target_power = 40) +Run: + ./rvs -c conf/MI355X/babel.conf -*) in case there are problems reaching/sustaining the given target_power +Configuration (`conf/MI355X/babel.conf`, first action): -**) please increase the ramp_interval and/or the tolerance value(s) and try again (in case of a 'ramp time exceeded' message) + actions: + - name: babel-double-825MiB + device: all + module: babel + parallel: false + count: 1 + num_iter: 2000 + duration: 0 + array_size: 865075200 + test_type: 2 + mibibytes: false + o/p_csv: false + read: true + write: true + copy: true + mul: true + add: true + dot: true + triad: true + dwords_per_lane: 4 + chunks_per_block: 1 + tb_size: 512 + +Sample output (first action, abridged): + + [RESULT] [302411.15197 ] Action name :babel-double-825MiB + [RESULT] [302411.96856 ] Module name :babel + [RESULT] [302411.735270] [babel-double-825MiB] [GPU:: 42583] Starting the Babel memory stress test + [RESULT] [302411.735337] Running kernels 2000 times, Precision: double + [RESULT] [302411.735378] Array size: 6920.6 MB (=6.9 GB), Total size: 20761.8 MB (=20.8 GB) + + [RESULT] [302446.903905] + --------------------------------------------------------------------------------- + GPU Id Function MBytes/sec Max MB/s Min MB/s Avg MB/s + --------------------------------------------------------------------------------- + 42583 Read 7090191.900 7090191.900 6638148.386 7019091.298 + 42583 Write 6645670.223 6645670.223 5618652.956 6106108.680 + 42583 Copy 6224488.986 6224488.986 6037686.546 6150313.849 + 42583 Mul 6238685.230 6238685.230 6056284.950 6152982.512 + 42583 Add 6008723.700 6008723.700 5485279.487 5919255.626 + 42583 Triad 6031416.066 6031416.066 5879289.313 5950804.403 + 42583 Dot 5844542.436 5844542.436 4912455.933 5806381.609 + --------------------------------------------------------------------------------- + ... -**) please increase the tolerance value (in case too many 'power violation message' are logged out) + +=====================================================================+ + | ROCm Validation Suite (RVS) Summary | + +=====================================================================+ + | System Overview | + +---------------------------------------------------------------------+ + | Operating System | Ubuntu 22.04.5 LTS | + | RVS version | 1.6.75 | + | ROCm version | 7.2.1-81 | + | amdgpu version | 6.16.13 | + | GPUs | 8 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 42583 | AMD Instinct MI355X - 27226 | + | 0 - 2 - 0000:05:00.0 | 1 - 3 - 0000:15:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 36479 | AMD Instinct MI355X - 17010 | + | 2 - 4 - 0000:65:00.0 | 3 - 5 - 0000:75:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 1590 | AMD Instinct MI355X - 51771 | + | 4 - 6 - 0000:85:00.0 | 5 - 7 - 0000:95:00.0 | + +---------------------------------------------------------------------+ + | AMD Instinct MI355X - 11806 | AMD Instinct MI355X - 57875 | + | 6 - 8 - 0000:e5:00.0 | 7 - 9 - 0000:f5:00.0 | + +=====================================================================+ + | Action Name | Module | Result | + +=====================================================================+ + | babel-double-825MiB | BABEL | PASS | + +---------------------------------------------------------------------+ diff --git a/docs/versions.md b/docs/versions.md new file mode 100644 index 000000000..834e24e22 --- /dev/null +++ b/docs/versions.md @@ -0,0 +1,19 @@ +:orphan: + + + + + + + +# ROCm RVS release history + +| Version | Release date | +| ------- | ------------ | +| [1.6](https://rocm.docs.amd.com/projects/ROCmValidationSuite/en/docs-1.6/) | August 26, 2026 | +| [1.5](https://rocm.docs.amd.com/projects/ROCmValidationSuite/en/docs-1.5/) | September 2, 2026 | +| [1.4.21](https://rocm.docs.amd.com/projects/ROCmValidationSuite/en/docs-1.4.21/) | May 15, 2026 | + +```{note} +RVS 1.4.21 is a technology preview release. For the latest production release, see the [RVS 1.6 documentation](https://rocm.docs.amd.com/projects/ROCmValidationSuite/en/latest/). +``` diff --git a/edp.so/CMakeLists.txt b/edp.so/CMakeLists.txt deleted file mode 100644 index 512d2aea7..000000000 --- a/edp.so/CMakeLists.txt +++ /dev/null @@ -1,181 +0,0 @@ -################################################################################ -## -## Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. -## -## MIT LICENSE: -## Permission is hereby granted, free of charge, to any person obtaining a copy of -## this software and associated documentation files (the "Software"), to deal in -## the Software without restriction, including without limitation the rights to -## use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies -## of the Software, and to permit persons to whom the Software is furnished to do -## so, subject to the following conditions: -## -## The above copyright notice and this permission notice shall be included in all -## copies or substantial portions of the Software. -## -## THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -## IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -## FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -## AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -## LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -## OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE -## SOFTWARE. -## -################################################################################ - -cmake_minimum_required ( VERSION 3.5.0 ) -if ( ${CMAKE_BINARY_DIR} STREQUAL ${CMAKE_CURRENT_SOURCE_DIR}) - message(FATAL "In-source build is not allowed") -endif () -set (CMAKE_RUNTIME_OUTPUT_DIRECTORY "${CMAKE_BINARY_DIR}/bin") - -set ( RVS "edp" ) -set ( RVS_PACKAGE "rvs-roct" ) -set ( RVS_COMPONENT "lib${RVS}" ) -set ( RVS_TARGET "${RVS}" ) - -project ( ${RVS_TARGET} ) - -message(STATUS "MODULE: ${RVS}") -add_compile_options(-Wall ) -if (RVS_COVERAGE) - add_compile_options(-o0 -fprofile-arcs -ftest-coverage) - set(CMAKE_EXE_LINKER_FLAGS "--coverage") - set(CMAKE_SHARED_LINKER_FLAGS "--coverage") -endif() - -## Set default module path if not already set -if ( NOT DEFINED CMAKE_MODULE_PATH ) - set ( CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/../cmake_modules/" ) -endif () - -## Include common cmake modules -include ( utils ) - -## Setup the package version. -get_version ( "0.0.0" ) - -set ( BUILD_VERSION_MAJOR ${VERSION_MAJOR} ) -set ( BUILD_VERSION_MINOR ${VERSION_MINOR} ) -set ( BUILD_VERSION_PATCH ${VERSION_PATCH} ) -set ( LIB_VERSION_STRING "${BUILD_VERSION_MAJOR}.${BUILD_VERSION_MINOR}.${BUILD_VERSION_PATCH}" ) - -if ( DEFINED VERSION_BUILD AND NOT ${VERSION_BUILD} STREQUAL "" ) - set ( BUILD_VERSION_PATCH "${BUILD_VERSION_PATCH}-${VERSION_BUILD}" ) -endif () -set ( BUILD_VERSION_STRING "${BUILD_VERSION_MAJOR}.${BUILD_VERSION_MINOR}.${BUILD_VERSION_PATCH}" ) - -## make version numbers visible to C code -add_compile_options(-DBUILD_VERSION_MAJOR=${VERSION_MAJOR}) -add_compile_options(-DBUILD_VERSION_MINOR=${VERSION_MINOR}) -add_compile_options(-DBUILD_VERSION_PATCH=${VERSION_PATCH}) -add_compile_options(-DLIB_VERSION_STRING="${LIB_VERSION_STRING}") -add_compile_options(-DBUILD_VERSION_STRING="${BUILD_VERSION_STRING}") - -set(ROCBLAS_LIB "rocblas") -set(HIP_HCC_LIB "amdhip64") - -#ROCBLAS VERSION CHECK FLAGS TO CHECK REORG VERSION 2.44.0 -add_compile_options(-DRVS_ROCBLAS_VERSION_FLAT=${RVS_ROCBLAS_VERSION_FLAT}) - -# Determine HSA_PATH -if(NOT DEFINED HIPCC_PATH) - if(NOT DEFINED ENV{HIPCC_PATH}) - set(HIPCC_PATH "${ROCM_PATH}" CACHE PATH "Path to which hipcc runtime has been installed") - else() - set(HIPCC_PATH $ENV{HIPCC_PATH} CACHE PATH "Path to which hipcc runtime has been installed") - endif() -endif() - -# Add HIP_VERSION to CMAKE__FLAGS -set(HIP_HCC_BUILD_FLAGS "${HIP_HCC_BUILD_FLAGS} -DHIP_VERSION_MAJOR=${HIP_VERSION_MAJOR} -DHIP_VERSION_MINOR=${HIP_VERSION_MINOR} -DHIP_VERSION_PATCH=${HIP_VERSION_GITDATE}") - -set(HIP_HCC_BUILD_FLAGS) -set(HIP_HCC_BUILD_FLAGS "${HIP_HCC_BUILD_FLAGS} -fPIC ${HCC_CXX_FLAGS} -I${HSA_INC_DIR}") - - -# Set compiler and compiler flags -set(CMAKE_CXX_COMPILER "${HIPCC_PATH}/bin/hipcc") -set(CMAKE_C_COMPILER "${HIPCC_PATH}/bin/hipcc") -set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} ${HIP_HCC_BUILD_FLAGS}") -set(CMAKE_C_FLAGS "${CMAKE_C_FLAGS} ${HIP_HCC_BUILD_FLAGS}") - -# Determine Roc Runtime header files are accessible -if(NOT EXISTS ${HIP_INC_DIR}/hip/hip_runtime.h) - message("ERROR: ROC Runtime headers can't be found under specified path. Please set HIP_INC_DIR path. Current value is : " ${HIP_INC_DIR}) - RETURN() -endif() - -if(NOT EXISTS ${HIP_INC_DIR}/hip/hip_runtime_api.h) - message("ERROR: ROC Runtime headers can't be found under specified path. Please set HIP_INC_DIR path. Current value is : " ${HIP_INC_DIR}) - RETURN() -endif() - -# Determine Roc Runtime header files are accessible -if(DEFINED RVS_ROCMSMI) - if(NOT RVS_ROCMSMI EQUAL 1) - if(NOT EXISTS ${ROCBLAS_INC_DIR}/${ROCBLAS_MODULE_NM_PREFIX}rocblas.h) - message("ERROR: rocBLAS headers can't be found under specified path. Please set ROCBLAS_INC_DIR path. Current value is : " ${ROCBLAS_INC_DIR}) - RETURN() - endif() - - if(NOT EXISTS "${ROCBLAS_LIB_DIR}/lib${ROCBLAS_LIB}.so") - message("ERROR: rocBLAS library can't be found under specified path. Please set ROCBLAS_LIB_DIR path. Current value is : " ${ROCBLAS_LIB_DIR}) - RETURN() - endif() - endif() -endif() - - -if(NOT EXISTS "${HIP_LIB_DIR}/lib${HIP_HCC_LIB}.so") - message("ERROR: ROC Runtime libraries can't be found under specified path. Please set HIP_LIB_DIR path. Current value is : " ${HIP_LIB_DIR}) - RETURN() -endif() - -## define include directories -include_directories(./ ../ ${ROCR_INC_DIR} ${ROCBLAS_INC_DIR} ${HIP_INC_DIR} ${YAML_CPP_INCLUDE_DIR} ${HIPRAND_INC_DIR} ${ROCRAND_INC_DIR}) -# Add directories to look for library files to link -link_directories(${RVS_LIB_DIR} ${ROCR_LIB_DIR} ${ROCBLAS_LIB_DIR} ${AMD_SMI_LIB_DIR} ${HIPRAND_LIB_DIR} ${ROCRAND_LIB_DIR}) -## additional libraries -set (PROJECT_LINK_LIBS rvslib libpthread.so libpciaccess.so libpci.so libm.so) - -## define source files -set (SOURCES src/rvs_module.cpp src/action.cpp src/edp_worker.cpp ) - -## define target -add_library( ${RVS_TARGET} SHARED ${SOURCES}) -set_target_properties(${RVS_TARGET} PROPERTIES - SUFFIX .so.${LIB_VERSION_STRING} - LIBRARY_OUTPUT_DIRECTORY ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}) -target_link_libraries(${RVS_TARGET} ${PROJECT_LINK_LIBS} ${HIP_HCC_LIB} ${ROCBLAS_LIB} ${HIPRAND_LIB} ${ROCRAND_LIB}) -add_dependencies(${RVS_TARGET} rvslib) - -add_custom_command(TARGET ${RVS_TARGET} POST_BUILD -COMMAND ln -fs ./lib${RVS}.so.${LIB_VERSION_STRING} lib${RVS}.so.${VERSION_MAJOR} WORKING_DIRECTORY ${CMAKE_RUNTIME_OUTPUT_DIRECTORY} -COMMAND ln -fs ./lib${RVS}.so.${VERSION_MAJOR} lib${RVS}.so WORKING_DIRECTORY ${CMAKE_RUNTIME_OUTPUT_DIRECTORY} -) - -add_custom_command( TARGET ${RVS_TARGET} POST_BUILD - COMMAND make clean - COMMAND make VERBOSE=1 - COMMAND cp -rf rocm_edp_helper ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/ - WORKING_DIRECTORY ${CMAKE_CURRENT_SOURCE_DIR}/rocm_edp_helper/ -) - -install(TARGETS ${RVS_TARGET} LIBRARY DESTINATION ${CPACK_PACKAGING_INSTALL_PREFIX}/rvs COMPONENT rvsmodule) -install( - FILES "${CMAKE_CURRENT_SOURCE_DIR}/rocm_edp_helper/rocm_edp_helper" - DESTINATION ${CPACK_PACKAGING_INSTALL_PREFIX}/${CMAKE_INSTALL_LIBDIR}/rvs COMPONENT rvsmodule - ) -install(FILES "${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/lib${RVS}.so.${VERSION_MAJOR}" - DESTINATION ${CPACK_PACKAGING_INSTALL_PREFIX}/${CMAKE_INSTALL_LIBDIR}/rvs COMPONENT rvsmodule) -install(FILES "${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/lib${RVS}.so" - DESTINATION ${CPACK_PACKAGING_INSTALL_PREFIX}/${CMAKE_INSTALL_LIBDIR}/rvs COMPONENT rvsmodule) - -# TEST SECTION -if (RVS_BUILD_TESTS) - add_custom_command(TARGET ${RVS_TARGET} POST_BUILD - COMMAND ln -fs ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/lib${RVS}.so.${VERSION_MAJOR} ${RVS_BINTEST_FOLDER}/lib${RVS}.so WORKING_DIRECTORY ${CMAKE_RUNTIME_OUTPUT_DIRECTORY} - ) - include(${CMAKE_CURRENT_SOURCE_DIR}/tests.cmake) -endif() diff --git a/edp.so/include/action.h b/edp.so/include/action.h deleted file mode 100644 index 8297e4466..000000000 --- a/edp.so/include/action.h +++ /dev/null @@ -1,128 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#ifndef EDP_SO_INCLUDE_ACTION_H_ -#define EDP_SO_INCLUDE_ACTION_H_ - -#ifdef __cplusplus -extern "C" { -#endif -#include -#ifdef __cplusplus -} -#endif - -#include -#include -#include - -#include "include/rvsactionbase.h" - -using std::vector; -using std::string; -using std::map; - -/** - * @class edp_action - * @ingroup EDP - * - * @brief EDP action implementation class - * - * Derives from rvs::actionbase and implements actual action functionality - * in its run() method. - * - */ -class edp_action: public rvs::actionbase { - public: - edp_action(); - virtual ~edp_action(); - - virtual int run(void); - - std::string edp_ops_type; - - protected: - - //! stress test ramp duration - uint64_t edp_ramp_interval; - //! maximum allowed number of target_stress violations - int edp_max_violations; - //! specifies whether to copy the matrices to the GPU before each - //! SGEMM operation - bool edp_copy_matrix; - //! target stress (in GFlops) that the GPU will try to achieve - float edp_target_stress; - //! GFlops tolerance (how much the GFlops can fluctuare after - //! the ramp period for the test to succeed) - float edp_tolerance; - - //Alpha and beta value - float edp_alpha_val; - float edp_beta_val; - - //! matrix size for SGEMM - uint64_t edp_matrix_size_a; - uint64_t edp_matrix_size_b; - uint64_t edp_matrix_size_c; - - //Parameter to heat up - uint64_t edp_hot_calls; - - //Tranpose set to none or enabled - int edp_trans_a; - int edp_trans_b; - - //Leading offset values - int edp_lda_offset; - int edp_ldb_offset; - int edp_ldc_offset; - - uint64_t edp_wave_iterations; - uint64_t edp_halt_timer; - uint64_t edp_restart_wave_timer; - bool edp_broadast_wave; - - // EDP specific config keys -// void property_get_edp_target_stress(int *error); -// void property_get_edp_tolerance(int *error); - - bool get_all_edp_config_keys(void); - /** - * @brief reads all common configuration keys from - * the module's properties collection - * @return true if no fatal error occured, false otherwise - */ - bool get_all_common_config_keys(void); - - /** - * @brief gets the number of ROCm compatible AMD GPUs - * @return run number of GPUs - */ - int get_num_amd_gpu_devices(void); - int get_all_selected_gpus(void); - bool do_gpu_stress_test(map edp_gpus_device_index); - void StartPeakPowerThread(unsigned int); -}; - -#endif // EDP_SO_INCLUDE_ACTION_H_ diff --git a/edp.so/include/edp_worker.h b/edp.so/include/edp_worker.h deleted file mode 100644 index 7db281ef6..000000000 --- a/edp.so/include/edp_worker.h +++ /dev/null @@ -1,275 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#ifndef EDP_SO_INCLUDE_EDP_WORKER_H_ -#define EDP_SO_INCLUDE_EDP_WORKER_H_ - -#include -#include -#include "include/rvsthreadbase.h" -#include "include/rvs_blas.h" - -#define EDP_RESULT_PASS_MESSAGE "true" -#define EDP_RESULT_FAIL_MESSAGE "false" - -/** - * @class EDPWorker - * @ingroup EDP - * - * @brief EDPWorker action implementation class - * - * Derives from rvs::ThreadBase and implements actual action functionality - * in its run() method. - * - */ -class EDPWorker : public rvs::ThreadBase { - public: - EDPWorker(); - virtual ~EDPWorker(); - - //! sets action name - void set_name(const std::string& name) { action_name = name; } - //! returns action name - const std::string& get_name(void) { return action_name; } - - //! sets GPU ID - void set_gpu_id(uint16_t _gpu_id) { gpu_id = _gpu_id; } - //! returns GPU ID - uint16_t get_gpu_id(void) { return gpu_id; } - - //! sets the GPU index - void set_gpu_device_index(int _gpu_device_index) { - gpu_device_index = _gpu_device_index; - } - //! returns the GPU index - int get_gpu_device_index(void) { return gpu_device_index; } - - //! sets the run delay - void set_run_wait_ms(uint64_t _run_wait_ms) { run_wait_ms = _run_wait_ms; } - //! returns the run delay - uint64_t get_run_wait_ms(void) { return run_wait_ms; } - - //! sets the total stress test run duration - void set_run_duration_ms(uint64_t _run_duration_ms) { - run_duration_ms = _run_duration_ms; - } - //! returns the total stress test run duration - uint64_t get_run_duration_ms(void) { return run_duration_ms; } - - //! sets the stress test ramp duration - void set_ramp_interval(uint64_t _ramp_interval) { - ramp_interval = _ramp_interval; - } - //! returns the stress test ramp duration - uint64_t get_ramp_interval(void) { return ramp_interval; } - - //! sets the time interval at which the module reports the average GFlops - void set_log_interval(uint64_t _log_interval) { - log_interval = _log_interval; - } - //! returns the time interval at which the module reports the average GFlops - uint64_t get_log_interval(void) { return log_interval; } - - //! sets the maximum allowed number of target_stress violations - void set_max_violations(uint64_t _max_violations) { - max_violations = _max_violations; - } - //! returns the maximum allowed number of target_stress violations - uint64_t get_max_violations(void) { return max_violations; } - - //! sets the copy_matrix (true = the matrix will be copied to GPU each - //! time a new GEMM will run, false = the matrix will be copied only once) - void set_copy_matrix(bool _copy_matrix) { copy_matrix = _copy_matrix; } - //! returns the copy_matrix value - bool get_copy_matrix(void) { return copy_matrix; } - - //! sets the target stress (in GFlops) that the GPU will try to achieve - void set_target_stress(float _target_stress) { - target_stress = _target_stress; - } - //! returns the target stress (in GFlops) that the GPU will try to achieve - float get_target_stress(void) { return target_stress; } - - //! sets hot calls - void set_edp_hot_calls(uint64_t _hot_calls) { - edp_hot_calls = _hot_calls; - } - - //! sets hot calls - uint64_t get_edp_hot_calls(void) { - return edp_hot_calls; - } - - //! sets the GEMM matrix size - void set_matrix_size_a(uint64_t _matrix_size_a) { - matrix_size_a = _matrix_size_a; - } - //! sets the GEMM matrix size - void set_matrix_size_b(uint64_t _matrix_size_b) { - matrix_size_b = _matrix_size_b; - } - //! sets the GEMM matrix size - void set_matrix_size_c(uint64_t _matrix_size_c) { - matrix_size_c = _matrix_size_c; - } - //! sets the transpose matrix a - void set_matrix_transpose_a(int transa) { - edp_trans_a = transa; - } - //! sets the transpose matrix b - void set_matrix_transpose_b(int transb) { - edp_trans_b = transb; - } - //! sets alpha val - void set_alpha_val(float alpha_val) { - edp_alpha_val = alpha_val; - } - //! sets beta val - void set_beta_val(float beta_val) { - edp_beta_val = beta_val; - } - - //! sets offsets - void set_lda_offset(int lda) { - edp_lda_offset = lda; - } - //! sets offsets - void set_ldb_offset(int ldb) { - edp_ldb_offset = ldb; - } - //! sets offsets - void set_ldc_offset(int ldc) { - edp_ldc_offset = ldc; - } - - void stopWaveInsideGPU(void ); - - - //! returns the GEMM matrix size - uint64_t get_matrix_size_a(void) { return matrix_size_a; } - - //! returns the GEMM matrix size - uint64_t get_matrix_size_b(void) { return matrix_size_b; } - - //! returns the GEMM matrix size - uint64_t get_matrix_size_c(void) { return matrix_size_b; } - - //! sets the GFlops tolerance - void set_tolerance(float _tolerance) { tolerance = _tolerance; } - //! returns the GFlops tolerance - float get_tolerance(void) { return tolerance; } - - - //! returns the difference (in milliseconds) between 2 points in time - uint64_t time_diff( - std::chrono::time_point t_end, - std::chrono::time_point t_start); - - //! sets the JSON flag - static void set_use_json(bool _bjson) { bjson = _bjson; } - //! returns the JSON flag - static bool get_use_json(void) { return bjson; } - - void set_edp_ops_type(std::string _ops_type) { edp_ops_type = _ops_type; } - - void set_wave_timer(int wavetimer) { edp_periodic_wave_timer = wavetimer; } - void set_halt_timer(int halttimer) { edp_halt_timer = halttimer; } - void set_restart_wave_timer(int restart_timer) { edp_restart_wave_timer = restart_timer; } - - protected: - void setup_blas(int *error, std::string *err_description); - void hit_max_gflops(int *error, std::string *err_description); - bool do_edp_ramp(int *error, std::string *err_description); - bool do_edp_stress_test(int *error, std::string *err_description); - void log_edp_test_result(bool edp_test_passed); - virtual void run(void); - void log_to_json(const std::string &key, const std::string &value, - int log_level); - void log_interval_gflops(double gflops_interval); - bool check_gflops_violation(double gflops_interval); - void check_target_stress(double gflops_interval); - void usleep_ex(uint64_t microseconds); - - protected: - //! name of the action - std::string action_name; - //! index of the GPU that will run the stress test - int gpu_device_index; - //Matrix transpose A - int edp_trans_a; - //Matrix transpose B - int edp_trans_b; - //! ID of the GPU that will run the stress test - uint16_t gpu_id; - //EDP aplha value - float edp_alpha_val; - //EDP beta value - float edp_beta_val; - //leading offsets - int edp_lda_offset; - int edp_ldb_offset; - int edp_ldc_offset; - //! stress test run delay - uint64_t run_wait_ms; - //! stress test run duration - uint64_t run_duration_ms; - //! stress test ramp duration - uint64_t ramp_interval; - //! time interval at which the module reports the average GFlops - uint64_t log_interval; - //! maximum allowed number of target_stress violations - uint64_t max_violations; - //! specifies whether to copy the matrix to the GPU for each GEMM operation - bool copy_matrix; - //! target stress (in GFlops) that the GPU will try to achieve - float target_stress; - //! GFlops tolerance (how much the GFlops can fluctuare after - //! the ramp period for the test to succeed) - float tolerance; - //! GEMM matrix size - uint64_t matrix_size_a; - uint64_t matrix_size_b; - uint64_t matrix_size_c; - - uint64_t edp_periodic_wave_timer; - uint64_t edp_halt_timer; - uint64_t edp_restart_wave_timer; - - //num of hot calls - uint64_t edp_hot_calls; - //! actual ramp time in case the GPU achieves the given target_stress Gflops - uint64_t ramp_actual_time; - //! rvs_blas pointer - std::unique_ptr gpu_blas; - //! max gflops achieved during the stress test - double max_gflops; - //! delay used to reduce GEMM frequency - double delay_target_stress; - //! TRUE if JSON output is required - static bool bjson; - //Type of operation - std::string edp_ops_type; -}; - -#endif // EDP_SO_INCLUDE_EDP_WORKER_H_ diff --git a/edp.so/src/.gitignore b/edp.so/src/.gitignore deleted file mode 100644 index 54d9ae01c..000000000 --- a/edp.so/src/.gitignore +++ /dev/null @@ -1 +0,0 @@ -/libmain.cpp diff --git a/edp.so/src/action.cpp b/edp.so/src/action.cpp deleted file mode 100644 index eacad634b..000000000 --- a/edp.so/src/action.cpp +++ /dev/null @@ -1,599 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#include "include/action.h" - -#include -#include -#include -#include -#include -#include -#include - -#define __HIP_PLATFORM_HCC__ -#include "hip/hip_runtime.h" -#include "hip/hip_runtime_api.h" - -#include "include/rvs_key_def.h" -#include "include/edp_worker.h" -#include "include/gpu_util.h" -#include "include/rvs_util.h" -#include "include/rvsactionbase.h" -#include "include/rvsloglp.h" - -extern "C" { - #include - #include -} - -using std::string; -using std::vector; -using std::map; -using std::regex; - -#define RVS_CONF_RAMP_INTERVAL_KEY "ramp_interval" -#define RVS_CONF_LOG_INTERVAL_KEY "log_interval" -#define RVS_CONF_MAX_VIOLATIONS_KEY "max_violations" -#define RVS_CONF_COPY_MATRIX_KEY "copy_matrix" -#define RVS_CONF_TARGET_STRESS_KEY "target_stress" -#define RVS_CONF_TOLERANCE_KEY "tolerance" -#define RVS_CONF_HOT_CALLS "hot_calls" -#define RVS_CONF_MATRIX_SIZE_KEYA "matrix_size_a" -#define RVS_CONF_MATRIX_SIZE_KEYB "matrix_size_b" -#define RVS_CONF_MATRIX_SIZE_KEYC "matrix_size_b" -#define RVS_CONF_EDP_OPS_TYPE "ops_type" -#define RVS_CONF_TRANS_A "transa" -#define RVS_CONF_TRANS_B "transb" -#define RVS_CONF_ALPHA_VAL "alpha" -#define RVS_CONF_BETA_VAL "beta" -#define RVS_CONF_LDA_OFFSET "lda" -#define RVS_CONF_LDB_OFFSET "ldb" -#define RVS_CONF_LDC_OFFSET "ldc" -#define RVS_CONF_HALT_WAVES "halt_wave_timer" -#define RVS_CONF_ITERATIONS "wave_iterations" -#define RVS_CONF_RESTART_WAVE_TIMER "restart_wave_timer" -#define RVS_CONF_BROADCAST_WAVE "broadcast" - -#define MODULE_NAME "edp" -#define MODULE_NAME_CAPS "EDP" - -#define EDP_DEFAULT_RAMP_INTERVAL 5000 -#define EDP_DEFAULT_LOG_INTERVAL 1000 -#define EDP_DEFAULT_MAX_VIOLATIONS 0 -#define EDP_DEFAULT_TOLERANCE 0.1 -#define EDP_DEFAULT_COPY_MATRIX true -#define EDP_DEFAULT_MATRIX_SIZE 5760 -#define EDP_DEFAULT_HOT_CALLS 0 -#define EDP_DEFAULT_TRANS_A 0 -#define EDP_DEFAULT_TRANS_B 1 -#define EDP_DEFAULT_ALPHA_VAL 1 -#define EDP_DEFAULT_BETA_VAL 1 -#define EDP_DEFAULT_LDA_OFFSET 0 -#define EDP_DEFAULT_LDB_OFFSET 0 -#define EDP_DEFAULT_LDC_OFFSET 0 -#define EDP_DEFAULT_HALT_WAVES 1000 -#define EDP_DEFAULT_WAVE_ITERATIONS 10000 -#define EDP_DEFAULT_RESTART_WAVE_TIMER 0 -#define EDP_DEFAULT_BROADCAST_WAVE false - -#define RVS_DEFAULT_PARALLEL false -#define RVS_DEFAULT_DURATION 0 - -#define EDP_NO_COMPATIBLE_GPUS "No AMD compatible GPU found!" - -#define FLOATING_POINT_REGEX "^[0-9]*\\.?[0-9]+$" - -#define JSON_CREATE_NODE_ERROR "JSON cannot create node" -#define EDP_DEFAULT_OPS_TYPE "sgemm" - -/** - * @brief default class constructor - */ -edp_action::edp_action() { - bjson = false; -} - -/** - * @brief class destructor - */ -edp_action::~edp_action() { - property.clear(); -} - - -/** - * @brief runs the EDP test stress session - * @param edp_gpus_device_index map - * @return true if no error occured, false otherwise - */ -bool edp_action::do_gpu_stress_test(map edp_gpus_device_index) { - size_t k = 0; - for (;;) { - unsigned int i = 0; - if (property_wait != 0) // delay edp execution - sleep(property_wait); - - vector workers(edp_gpus_device_index.size()); - - map::iterator it; - - // all worker instances have the same json settings - EDPWorker::set_use_json(bjson); - - for (it = edp_gpus_device_index.begin(); - it != edp_gpus_device_index.end(); ++it) { - // set worker thread stress test params - workers[i].set_name(action_name); - workers[i].set_gpu_id(it->second); - workers[i].set_gpu_device_index(it->first); - workers[i].set_run_wait_ms(property_wait); - workers[i].set_run_duration_ms(property_duration); - workers[i].set_ramp_interval(edp_ramp_interval); - workers[i].set_log_interval(property_log_interval); - workers[i].set_max_violations(edp_max_violations); - workers[i].set_copy_matrix(edp_copy_matrix); - workers[i].set_target_stress(edp_target_stress); - workers[i].set_tolerance(edp_tolerance); - workers[i].set_edp_hot_calls(edp_hot_calls); - workers[i].set_matrix_size_a(edp_matrix_size_a); - workers[i].set_matrix_size_b(edp_matrix_size_b); - workers[i].set_matrix_size_c(edp_matrix_size_c); - workers[i].set_edp_ops_type(edp_ops_type); - workers[i].set_matrix_transpose_a(edp_trans_a); - workers[i].set_matrix_transpose_b(edp_trans_b); - workers[i].set_alpha_val(edp_alpha_val); - workers[i].set_beta_val(edp_beta_val); - workers[i].set_lda_offset(edp_lda_offset); - workers[i].set_ldb_offset(edp_ldb_offset); - workers[i].set_ldc_offset(edp_ldc_offset); - workers[i].set_wave_timer(edp_wave_iterations); - workers[i].set_halt_timer(edp_halt_timer); - workers[i].set_restart_wave_timer(edp_restart_wave_timer); - - i++; - } - - if (property_parallel) { - for (i = 0; i < edp_gpus_device_index.size(); i++) - workers[i].start(); - - // join threads - for (i = 0; i < edp_gpus_device_index.size(); i++) - workers[i].join(); - } else { - for (i = 0; i < edp_gpus_device_index.size(); i++) { - workers[i].start(); - workers[i].join(); - - // check if stop signal was received - if (rvs::lp::Stopping()) - return false; - } - } - - // check if stop signal was received - if (rvs::lp::Stopping()) - return false; - - if (property_count != 0) { - k++; - if (k == property_count) - break; - } - } - - return rvs::lp::Stopping() ? false : true; -} - -/** - * @brief reads all EDP-related configuration keys from - * the module's properties collection - * @return true if no fatal error occured, false otherwise - */ -bool edp_action::get_all_edp_config_keys(void) { - int error; - string msg, ststress; - bool bsts = true; - - if ((error = - property_get(RVS_CONF_TARGET_STRESS_KEY, &edp_target_stress))) { - switch (error) { // is mandatory => EDP cannot continue - case 1: - msg = "invalid '" + std::string(RVS_CONF_TARGET_STRESS_KEY) + - "' key value " + ststress; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - break; - - case 2: - msg = "key '" + std::string(RVS_CONF_TARGET_STRESS_KEY) + - "' was not found"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - } - bsts = false; - } - - if (property_get_int(RVS_CONF_RAMP_INTERVAL_KEY, - &edp_ramp_interval, EDP_DEFAULT_RAMP_INTERVAL)) { - msg = "invalid '" + - std::string(RVS_CONF_RAMP_INTERVAL_KEY) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - if (property_get_int(RVS_CONF_LOG_INTERVAL_KEY, - &property_log_interval, EDP_DEFAULT_LOG_INTERVAL)) { - msg = "invalid '" + - std::string(RVS_CONF_LOG_INTERVAL_KEY) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - if (property_get_int(RVS_CONF_MAX_VIOLATIONS_KEY, &edp_max_violations, - EDP_DEFAULT_MAX_VIOLATIONS)) { - msg = "invalid '" + - std::string(RVS_CONF_MAX_VIOLATIONS_KEY) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - if (property_get(RVS_CONF_COPY_MATRIX_KEY, &edp_copy_matrix, - EDP_DEFAULT_COPY_MATRIX)) { - msg = "invalid '" + - std::string(RVS_CONF_COPY_MATRIX_KEY) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - if (property_get(RVS_CONF_TOLERANCE_KEY, &edp_tolerance, - EDP_DEFAULT_TOLERANCE)) { - msg = "invalid '" + - std::string(RVS_CONF_TOLERANCE_KEY) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - if (property_get(RVS_CONF_EDP_OPS_TYPE, &edp_ops_type, - EDP_DEFAULT_OPS_TYPE)) { - msg = "invalid '" + - std::string(RVS_CONF_EDP_OPS_TYPE) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - error = property_get_int(RVS_CONF_HOT_CALLS, &edp_hot_calls, EDP_DEFAULT_HOT_CALLS); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_HOT_CALLS) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - - error = property_get_int(RVS_CONF_MATRIX_SIZE_KEYA, &edp_matrix_size_a, EDP_DEFAULT_MATRIX_SIZE); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_MATRIX_SIZE_KEYA) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - error = property_get_int(RVS_CONF_MATRIX_SIZE_KEYB, &edp_matrix_size_b, EDP_DEFAULT_MATRIX_SIZE); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_MATRIX_SIZE_KEYB) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - error = property_get_int(RVS_CONF_MATRIX_SIZE_KEYC, &edp_matrix_size_c, EDP_DEFAULT_MATRIX_SIZE); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_MATRIX_SIZE_KEYC) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - error = property_get_int(RVS_CONF_TRANS_A, &edp_trans_a, EDP_DEFAULT_TRANS_A); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_TRANS_A) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - error = property_get_int(RVS_CONF_TRANS_B, &edp_trans_b, EDP_DEFAULT_TRANS_B); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_TRANS_B) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - error = property_get(RVS_CONF_ALPHA_VAL, &edp_alpha_val, EDP_DEFAULT_ALPHA_VAL); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_ALPHA_VAL) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - error = property_get(RVS_CONF_BETA_VAL, &edp_beta_val, EDP_DEFAULT_BETA_VAL); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_BETA_VAL) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - error = property_get_int(RVS_CONF_LDA_OFFSET, &edp_lda_offset, EDP_DEFAULT_LDA_OFFSET); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_LDA_OFFSET) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - error = property_get_int(RVS_CONF_LDB_OFFSET, &edp_ldb_offset, EDP_DEFAULT_LDB_OFFSET); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_LDB_OFFSET) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - error = property_get_int(RVS_CONF_LDC_OFFSET, &edp_ldc_offset, EDP_DEFAULT_LDC_OFFSET); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_LDC_OFFSET) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - error = property_get_int(RVS_CONF_ITERATIONS, &edp_wave_iterations, EDP_DEFAULT_WAVE_ITERATIONS); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_ITERATIONS) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - error = property_get_int(RVS_CONF_HALT_WAVES, &edp_halt_timer, EDP_DEFAULT_HALT_WAVES); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_HALT_WAVES) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - error = property_get_int(RVS_CONF_RESTART_WAVE_TIMER, &edp_restart_wave_timer, EDP_DEFAULT_RESTART_WAVE_TIMER); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_RESTART_WAVE_TIMER) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - error = property_get(RVS_CONF_BROADCAST_WAVE, &edp_broadast_wave, EDP_DEFAULT_BROADCAST_WAVE); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_BROADCAST_WAVE) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - - - - return bsts; -} - -/** - * @brief reads all common configuration keys from - * the module's properties collection - * @return true if no fatal error occured, false otherwise - */ -bool edp_action::get_all_common_config_keys(void) { - string msg, sdevid, sdev; - int error; - bool bsts = true; - - // get property value (a list of gpu id) - if (int sts = property_get_device()) { - switch (sts) { - case 1: - msg = "Invalid 'device' key value."; - break; - case 2: - msg = "Missing 'device' key."; - break; - } - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - // get the property value if provided - if (property_get_int(RVS_CONF_DEVICEID_KEY, - &property_device_id, 0u)) { - msg = "Invalid 'deviceid' key value."; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - // get property value (a list of device indexes) - if (int sts = property_get_device_index()) { - switch (sts) { - case 1: - msg = "Invalid 'device_index' key value."; - break; - case 2: - msg = "Missing 'device_index' key."; - break; - } - // default set as true - property_device_index_all = true; - rvs::lp::Log(msg, rvs::loginfo); - } - - // get the other action/EDP related properties - if (property_get(RVS_CONF_PARALLEL_KEY, &property_parallel, false)) { - msg = "invalid '" + - std::string(RVS_CONF_PARALLEL_KEY) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - error = property_get_int - (RVS_CONF_COUNT_KEY, &property_count, DEFAULT_COUNT); - if (error != 0) { - msg = "invalid '" + - std::string(RVS_CONF_COUNT_KEY) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - error = property_get_int - (RVS_CONF_WAIT_KEY, &property_wait, DEFAULT_WAIT); - if (error != 0) { - msg = "invalid '" + - std::string(RVS_CONF_WAIT_KEY) + "' key value"; - bsts = false; - } - - error = property_get_int - (RVS_CONF_DURATION_KEY, &property_duration, RVS_DEFAULT_DURATION); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_DURATION_KEY) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - return bsts; -} - -/** - * @brief gets the number of ROCm compatible AMD GPUs - * @return run number of GPUs - */ -int edp_action::get_num_amd_gpu_devices(void) { - int hip_num_gpu_devices; - string msg; - - hipGetDeviceCount(&hip_num_gpu_devices); - if (hip_num_gpu_devices == 0) { // no AMD compatible GPU - msg = action_name + " " + MODULE_NAME + " " + EDP_NO_COMPATIBLE_GPUS; - rvs::lp::Log(msg, rvs::logerror); - - if (bjson) { - unsigned int sec; - unsigned int usec; - rvs::lp::get_ticks(&sec, &usec); - void *json_root_node = rvs::lp::LogRecordCreate(MODULE_NAME, - action_name.c_str(), rvs::loginfo, sec, usec); - if (!json_root_node) { - // log the error - string msg = std::string(JSON_CREATE_NODE_ERROR); - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - return -1; - } - - rvs::lp::AddString(json_root_node, "ERROR", EDP_NO_COMPATIBLE_GPUS); - rvs::lp::LogRecordFlush(json_root_node); - } - return 0; - } - return hip_num_gpu_devices; -} - - -/** - * @brief gets all selected GPUs and starts the worker threads - * @return run result - */ -int edp_action::get_all_selected_gpus(void) { - int hip_num_gpu_devices; - bool amd_gpus_found = false; - map edp_gpus_device_index; - std::string msg; - char buff[75]; - uint32_t iterations = 0; - - hip_num_gpu_devices = get_num_amd_gpu_devices(); - if (hip_num_gpu_devices < 1) - return hip_num_gpu_devices; - - //system("./rocm_edp_helper -l 1000000 &"); - //system(sprintf("./rocm_edp_helper -l %d &", edp_wave_iterations)); - sprintf(buff, "./rocm_edp_helper -l %d &", edp_wave_iterations); - system(buff); - - // iterate over all available & compatible AMD GPUs - amd_gpus_found = fetch_gpu_list(hip_num_gpu_devices, edp_gpus_device_index, - property_device, property_device_id, property_device_all); - if (amd_gpus_found) { - if (do_gpu_stress_test(edp_gpus_device_index)) - return 0; - - return -1; - } else { - msg = "No devices match criteria from the test configuration."; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - return -1; - } - - return 0; -} - -/** - * @brief runs the whole EDP logic - * @return run result - */ -int edp_action::run(void) { - string msg; - - // get the action name - if (property_get(RVS_CONF_NAME_KEY, &action_name)) { - rvs::lp::Err("Action name missing", MODULE_NAME_CAPS); - return -1; - } - - // check for -j flag (json logging) - if (property.find("cli.-j") != property.end()) - bjson = true; - - if (!get_all_common_config_keys()) - return -1; - if (!get_all_edp_config_keys()) - return -1; - - if (property_duration > 0 && (property_duration < edp_ramp_interval)) { - msg = "'" + - std::string(RVS_CONF_DURATION_KEY) + "' cannot be less than '" + - std::string(RVS_CONF_RAMP_INTERVAL_KEY) + "'"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - return -1; - } - - return get_all_selected_gpus(); -} diff --git a/edp.so/src/edp_worker.cpp b/edp.so/src/edp_worker.cpp deleted file mode 100644 index 59b6ede4c..000000000 --- a/edp.so/src/edp_worker.cpp +++ /dev/null @@ -1,355 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#include "include/edp_worker.h" - -#include -#include -#include -#include -#include - -#include "include/rvs_blas.h" -#include "include/rvs_module.h" -#include "include/rvsloglp.h" -#include "include/rvstimer.h" - -extern "C" { - #include - #include -} - -#define MODULE_NAME "edp" - -#define EDP_MEM_ALLOC_ERROR "memory allocation error!" -#define EDP_BLAS_ERROR "memory/blas error!" -#define EDP_BLAS_MEMCPY_ERROR "HostToDevice mem copy error!" - -#define EDP_MAX_GFLOPS_OUTPUT_KEY "Gflop" -#define EDP_FLOPS_PER_OP_OUTPUT_KEY "flops_per_op" -#define EDP_BYTES_COPIED_PER_OP_OUTPUT_KEY "bytes_copied_per_op" -#define EDP_TRY_OPS_PER_SEC_OUTPUT_KEY "try_ops_per_sec" - -#define EDP_LOG_GFLOPS_INTERVAL_KEY "Gflops" -#define EDP_JSON_LOG_GPU_ID_KEY "gpu_id" - -#define PROC_DEC_INC_SGEMM_FREQ_DELAY 10 - -#define NMAX_MS_GPU_RUN_PEAK_PERFORMANCE 1000 -#define NMAX_MS_SGEMM_OPS_RAMP_SUB_INTERVAL 1000 -#define USLEEP_MAX_VAL (1000000 - 1) - -#define EDP_COPY_MATRIX_MSG "copy matrix" -#define EDP_START_MSG "start" -#define EDP_PASS_KEY "pass" -#define EDP_RAMP_EXCEEDED_MSG "ramp time exceeded" -#define EDP_TARGET_ACHIEVED_MSG "target achieved" -#define EDP_STRESS_VIOLATION_MSG "stress violation" - -using std::string; - -bool EDPWorker::bjson = false; -static std::atomic flag(false); - -EDPWorker::EDPWorker() {} -EDPWorker::~EDPWorker() {} - -/** - * @brief performs the rvsBlas setup - * @param error pointer to a memory location where the error code will be stored - * @param err_description stores the error description if any - */ -void EDPWorker::setup_blas(int *error, string *err_description) { - *error = 0; - // setup rvsBlas - gpu_blas = std::unique_ptr( - new rvs_blas(gpu_device_index, matrix_size_a, matrix_size_b, - matrix_size_c, edp_trans_a, edp_trans_b, - edp_alpha_val, edp_beta_val, - edp_lda_offset, edp_ldb_offset, edp_ldc_offset, edp_ops_type)); - - if (!gpu_blas) { - *error = 1; - *err_description = EDP_MEM_ALLOC_ERROR; - return; - } - - if (gpu_blas->error()) { - *error = 1; - *err_description = EDP_MEM_ALLOC_ERROR; - return; - } - - // generate random matrix & copy it to the GPU - gpu_blas->generate_random_matrix_data(); - if (!copy_matrix) { - // copy matrix only once - if (!gpu_blas->copy_data_to_gpu(edp_ops_type)) { - *error = 1; - *err_description = EDP_BLAS_MEMCPY_ERROR; - } - } -} - -/** - * @brief logs the Gflops computed over the last log_interval period - * @param gflops_interval the Gflops that the GPU achieved - */ -void EDPWorker::check_target_stress(double gflops_interval) { - string msg; - bool result; - - if(gflops_interval >= target_stress){ - result = true; - }else{ - result = false; - } - - msg = "[" + action_name + "] " + MODULE_NAME + " " + - std::to_string(gpu_id) + " " + EDP_LOG_GFLOPS_INTERVAL_KEY + " " + std::to_string(gflops_interval) + " " + - "Target stress :" + " " + std::to_string(target_stress) + " met :" + (result ? "TRUE" : "FALSE"); - rvs::lp::Log(msg, rvs::logresults); - - log_to_json(EDP_LOG_GFLOPS_INTERVAL_KEY, std::to_string(gflops_interval), - rvs::loginfo); -} - - - -/** - * @brief logs the Gflops computed over the last log_interval period - * @param gflops_interval the Gflops that the GPU achieved - */ -void EDPWorker::log_interval_gflops(double gflops_interval) { - string msg; - msg = "[" + action_name + "] " + MODULE_NAME + " " + - std::to_string(gpu_id) + " " + EDP_LOG_GFLOPS_INTERVAL_KEY + " " + - std::to_string(gflops_interval); - rvs::lp::Log(msg, rvs::loginfo); - - log_to_json(EDP_LOG_GFLOPS_INTERVAL_KEY, std::to_string(gflops_interval), - rvs::loginfo); -} - - - - -/** - * @brief performs the stress test on the given GPU - * @param error pointer to a memory location where the error code will be stored - * @param err_description stores the error description if any - * @return true if stress violations is less than max_violations, false otherwise - */ -bool EDPWorker::do_edp_stress_test(int *error, std::string *err_description) { - uint16_t num_sgemm_ops = 0; - uint16_t num_gflops_violations = 0; - uint64_t total_milliseconds, log_interval_milliseconds; - uint64_t start_time, end_time; - double seconds_elapsed, gflops_interval; - double timetakenforoneiteration; - string msg; - std::chrono::time_point edp_start_time, - edp_end_time, edp_log_interval_time; - - *error = 0; - max_gflops = 0; - num_sgemm_ops = 0; - start_time = 0; - end_time = 0; - - edp_start_time = std::chrono::system_clock::now(); - edp_log_interval_time = std::chrono::system_clock::now(); - - // setup rvs blas - setup_blas(error, err_description); - if (*error) - return false; - - for (;;) { - - //Start the timer - start_time = gpu_blas->get_time_us(); - - // run GEMM & wait for completion - gpu_blas->run_blass_gemm(edp_ops_type); - - //End the timer - end_time = gpu_blas->get_time_us(); - - //Converting microseconds to seconds - timetakenforoneiteration = (end_time - start_time)/1e6; - - gflops_interval = gpu_blas->gemm_gflop_count()/timetakenforoneiteration; - - log_interval_gflops(gflops_interval); - - if(edp_hot_calls == 0) { - break; - }else{ - edp_hot_calls--; - } - - } - - return true; -} - - -/** - * @brief performs the stress test on the given GPU - */ -void EDPWorker::run() { - //pthread_t thread; - string err_description; - string msg; - bool edp_test_passed; - int interval; - int error; - - edp_test_passed = true; - interval = edp_periodic_wave_timer; - max_gflops = 0; - error = 0; - - //pthread_create(&thread, NULL, enable_disable_waves, &interval); - - // log EDP stress test - start message - msg = "[" + action_name + "] " + MODULE_NAME + " " + - std::to_string(gpu_id) + " " + EDP_START_MSG + " " + - " Starting the EDP stress test "; - rvs::lp::Log(msg, rvs::logtrace); - - log_to_json(EDP_START_MSG, std::to_string(target_stress), rvs::loginfo); - log_to_json(EDP_COPY_MATRIX_MSG, (copy_matrix ? "true":"false"), - rvs::loginfo); - - if (run_duration_ms > 0) { - edp_test_passed = do_edp_stress_test(&error, &err_description); - // check if stop signal was received - if (rvs::lp::Stopping()) - return; - - if (error) { - // GPU didn't complete the test (HIP/rocBlas error(s) occurred) - string msg = "[" + action_name + "] " + MODULE_NAME + " " + - std::to_string(gpu_id) + " " + err_description; - rvs::lp::Log(msg, rvs::logerror); - log_to_json("err", err_description, rvs::logerror); - return; - } - } - - log_interval_gflops(max_gflops); -} - -/** - * @brief logs the EDP test result - * @param edp_test_passed true if test succeeded, false otherwise - */ -void EDPWorker::log_edp_test_result(bool edp_test_passed) { - string msg; - - double flops_per_op = (2 * (static_cast(gpu_blas->get_m())/1000) * - (static_cast(gpu_blas->get_n())/1000) * - (static_cast(gpu_blas->get_k())/1000)); - msg = "[" + action_name + "] " + MODULE_NAME + " " + - std::to_string(gpu_id) + " " + EDP_MAX_GFLOPS_OUTPUT_KEY + ": " + - std::to_string(max_gflops) + " " + EDP_FLOPS_PER_OP_OUTPUT_KEY + ": " + - std::to_string(flops_per_op) + "x1e9" + " " + - EDP_BYTES_COPIED_PER_OP_OUTPUT_KEY + ": " + - std::to_string(gpu_blas->get_bytes_copied_per_op()) + - " " + EDP_TRY_OPS_PER_SEC_OUTPUT_KEY + ": "+ - std::to_string(target_stress / gpu_blas->gemm_gflop_count()) + - " " ; - rvs::lp::Log(msg, rvs::logresults); - - log_to_json(EDP_MAX_GFLOPS_OUTPUT_KEY, std::to_string(max_gflops), - rvs::loginfo); - log_to_json(EDP_FLOPS_PER_OP_OUTPUT_KEY, std::to_string(flops_per_op) + - "x1e9", rvs::loginfo); - log_to_json(EDP_BYTES_COPIED_PER_OP_OUTPUT_KEY, - std::to_string(gpu_blas->get_bytes_copied_per_op()), - rvs::loginfo); - log_to_json(EDP_TRY_OPS_PER_SEC_OUTPUT_KEY, - std::to_string(target_stress / gpu_blas->gemm_gflop_count()), - rvs::loginfo); - log_to_json(EDP_PASS_KEY, (edp_test_passed ? - EDP_RESULT_PASS_MESSAGE : EDP_RESULT_FAIL_MESSAGE), - rvs::logresults); -} - -/** - * @brief computes the difference (in milliseconds) between 2 points in time - * @param t_end second point in time - * @param t_start first point in time - * @return time difference in milliseconds - */ -uint64_t EDPWorker::time_diff( - std::chrono::time_point t_end, - std::chrono::time_point t_start) { - auto milliseconds = std::chrono::duration_cast( - t_end - t_start); - return milliseconds.count(); -} - -/** - * @brief logs a message to JSON - * @param key info type - * @param value message to log - * @param log_level the level of log (e.g.: info, results, error) - */ -void EDPWorker::log_to_json(const std::string &key, const std::string &value, - int log_level) { - if (EDPWorker::bjson) { - unsigned int sec; - unsigned int usec; - - rvs::lp::get_ticks(&sec, &usec); - void *json_node = rvs::lp::LogRecordCreate(MODULE_NAME, - action_name.c_str(), log_level, sec, usec); - if (json_node) { - rvs::lp::AddString(json_node, EDP_JSON_LOG_GPU_ID_KEY, - std::to_string(gpu_id)); - rvs::lp::AddString(json_node, key, value); - rvs::lp::LogRecordFlush(json_node); - } - } -} - -/** - * @brief extends the usleep for more than 1000000us - * @param microseconds us to sleep - */ -void EDPWorker::usleep_ex(uint64_t microseconds) { - uint64_t total_microseconds = microseconds; - for (;;) { - if (total_microseconds > USLEEP_MAX_VAL) { - usleep(USLEEP_MAX_VAL); - total_microseconds -= USLEEP_MAX_VAL; - } else { - usleep(total_microseconds); - return; - } - } -} diff --git a/edp.so/src/rvs_module.cpp b/edp.so/src/rvs_module.cpp deleted file mode 100644 index 39e1a3542..000000000 --- a/edp.so/src/rvs_module.cpp +++ /dev/null @@ -1,94 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#include "include/rvs_module.h" -#include "include/action.h" -#include "include/rvsloglp.h" -#include "include/gpu_util.h" - -/** - * @defgroup EDP EDP Module - * - * @brief performs GPU Stress Test - * - * The GPU Stress Test runs a Graphics Stress test or SGEMM/DGEMM - * (Single/Double-precision General Matrix Multiplication) workload - * on one, some or all GPUs. The GPUs can be of the same or different types. - * The duration of the benchmark should be configurable, both in terms of time - * (how long to run) and iterations (how many times to run). - * - */ - -extern "C" int rvs_module_has_interface(int iid) { - int sts = 0; - switch (iid) { - case 0: - case 1: - sts = 1; - } - return sts; -} - -extern "C" const char* rvs_module_get_description(void) { - return "ROCm Validation Suite EDP module"; -} - -extern "C" const char* rvs_module_get_config(void) { - return "target_stress (float), copy_matrix (bool), "\ - "ramp_interval (int), tolerance (float), "\ - "max_violations (int), log_interval (int), "\ - "matrix_size (int)"; -} - -extern "C" const char* rvs_module_get_output(void) { - return "pass (bool)"; -} - -extern "C" int rvs_module_init(void* pMi) { - rvs::lp::Initialize(static_cast(pMi)); - rvs::gpulist::Initialize(); - return 0; -} - -extern "C" int rvs_module_terminate(void) { - return 0; -} - -extern "C" void* rvs_module_action_create(void) { - return static_cast(new edp_action); -} - -extern "C" int rvs_module_action_destroy(void* pAction) { - delete static_cast(pAction); - return 0; -} - -extern "C" int rvs_module_action_property_set(void* pAction, const char* Key, - const char* Val) { - return static_cast(pAction)->property_set(Key, Val); -} - -extern "C" int rvs_module_action_run(void* pAction) { - return static_cast(pAction)->run(); -} diff --git a/edp.so/tests.cmake b/edp.so/tests.cmake deleted file mode 100644 index 84e6231fe..000000000 --- a/edp.so/tests.cmake +++ /dev/null @@ -1,27 +0,0 @@ -################################################################################ -## -## Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. -## -## MIT LICENSE: -## Permission is hereby granted, free of charge, to any person obtaining a copy of -## this software and associated documentation files (the "Software"), to deal in -## the Software without restriction, including without limitation the rights to -## use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies -## of the Software, and to permit persons to whom the Software is furnished to do -## so, subject to the following conditions: -## -## The above copyright notice and this permission notice shall be included in all -## copies or substantial portions of the Software. -## -## THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -## IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -## FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -## AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -## LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -## OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE -## SOFTWARE. -## -################################################################################ - - -include(tests_conf_logging) diff --git a/external/TransferBench b/external/TransferBench new file mode 160000 index 000000000..5fbfa95a1 --- /dev/null +++ b/external/TransferBench @@ -0,0 +1 @@ +Subproject commit 5fbfa95a1111a7f33f20b9780e5f22bf739f82fa diff --git a/gm.so/.gitignore b/gm.so/.gitignore deleted file mode 100644 index 49f0bb943..000000000 --- a/gm.so/.gitignore +++ /dev/null @@ -1,9 +0,0 @@ -/.settings/ -/CMakeFiles/ -/Debug/ -/build/ -/cmake_install.cmake -/Makefile -/.project -/lib*.so.* - diff --git a/gm.so/CMakeLists.txt b/gm.so/CMakeLists.txt deleted file mode 100644 index 9d1ab2cef..000000000 --- a/gm.so/CMakeLists.txt +++ /dev/null @@ -1,155 +0,0 @@ -################################################################################ -## -## Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. -## -## MIT LICENSE: -## Permission is hereby granted, free of charge, to any person obtaining a copy of -## this software and associated documentation files (the "Software"), to deal in -## the Software without restriction, including without limitation the rights to -## use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies -## of the Software, and to permit persons to whom the Software is furnished to do -## so, subject to the following conditions: -## -## The above copyright notice and this permission notice shall be included in all -## copies or substantial portions of the Software. -## -## THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -## IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -## FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -## AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -## LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -## OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE -## SOFTWARE. -## -################################################################################ - -cmake_minimum_required ( VERSION 3.5.0 ) -if ( ${CMAKE_BINARY_DIR} STREQUAL ${CMAKE_CURRENT_SOURCE_DIR}) - message(FATAL "In-source build is not allowed") -endif () -set (CMAKE_RUNTIME_OUTPUT_DIRECTORY "${CMAKE_BINARY_DIR}/bin") - -set ( RVS "gm" ) -set ( RVS_PACKAGE "rvs-roct" ) -set ( RVS_COMPONENT "lib${RVS}" ) -set ( RVS_TARGET "${RVS}" ) - -project ( ${RVS_TARGET} ) - -message(STATUS "MODULE: ${RVS}") - -add_compile_options(-pthread) -add_compile_options(-Wall ) - -if (RVS_COVERAGE) - add_compile_options(-o0 -fprofile-arcs -ftest-coverage) - set(CMAKE_EXE_LINKER_FLAGS "--coverage") - set(CMAKE_SHARED_LINKER_FLAGS "--coverage") -endif() - -## Set default module path if not already set -if ( NOT DEFINED CMAKE_MODULE_PATH ) - set ( CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/../cmake_modules/" ) -endif () - -## Include common cmake modules -include ( utils ) - -## Setup the package version. -get_version ( "0.0.0" ) - -set ( BUILD_VERSION_MAJOR ${VERSION_MAJOR} ) -set ( BUILD_VERSION_MINOR ${VERSION_MINOR} ) -set ( BUILD_VERSION_PATCH ${VERSION_PATCH} ) -set ( LIB_VERSION_STRING "${BUILD_VERSION_MAJOR}.${BUILD_VERSION_MINOR}.${BUILD_VERSION_PATCH}" ) - -if ( DEFINED VERSION_BUILD AND NOT ${VERSION_BUILD} STREQUAL "" ) - set ( BUILD_VERSION_PATCH "${BUILD_VERSION_PATCH}-${VERSION_BUILD}" ) -endif () -set ( BUILD_VERSION_STRING "${BUILD_VERSION_MAJOR}.${BUILD_VERSION_MINOR}.${BUILD_VERSION_PATCH}" ) - -## make version numbers visible to C code -add_compile_options(-DBUILD_VERSION_MAJOR=${VERSION_MAJOR}) -add_compile_options(-DBUILD_VERSION_MINOR=${VERSION_MINOR}) -add_compile_options(-DBUILD_VERSION_PATCH=${VERSION_PATCH}) -add_compile_options(-DLIB_VERSION_STRING="${LIB_VERSION_STRING}") -add_compile_options(-DBUILD_VERSION_STRING="${BUILD_VERSION_STRING}") - - -# Determine HSA_PATH -if(NOT DEFINED HIPCC_PATH) - if(NOT DEFINED ENV{HIPCC_PATH}) - set(HIPCC_PATH "${ROCM_PATH}" CACHE PATH "Path to which hipcc runtime has been installed") - else() - set(HIPCC_PATH $ENV{HIPCC_PATH} CACHE PATH "Path to which hipcc runtime has been installed") - endif() -endif() - -# Add HIP_VERSION to CMAKE__FLAGS -set(HIP_HCC_BUILD_FLAGS "${HIP_HCC_BUILD_FLAGS} -DHIP_VERSION_MAJOR=${HIP_VERSION_MAJOR} -DHIP_VERSION_MINOR=${HIP_VERSION_MINOR} -DHIP_VERSION_PATCH=${HIP_VERSION_GITDATE}") - -set(HIP_HCC_BUILD_FLAGS) -set(HIP_HCC_BUILD_FLAGS "${HIP_HCC_BUILD_FLAGS} -fPIC ${HCC_CXX_FLAGS} -I${HSA_INC_DIR} ${ASAN_CXX_FLAGS}") - -# Set compiler and compiler flags -set(CMAKE_CXX_COMPILER "${HIPCC_PATH}/bin/hipcc") -set(CMAKE_C_COMPILER "${HIPCC_PATH}/bin/hipcc") -set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} ${HIP_HCC_BUILD_FLAGS}") -set(CMAKE_C_FLAGS "${CMAKE_C_FLAGS} ${HIP_HCC_BUILD_FLAGS}") -set(CMAKE_EXE_LINKER_FLAGS "${CMAKE_EXE_LINKER_FLAGS} ${ASAN_LD_FLAGS}") -set(CMAKE_SHARED_LINKER_FLAGS "${CMAKE_SHARED_LINKER_FLAGS} ${ASAN_LD_FLAGS}") - -if(BUILD_ADDRESS_SANITIZER) - execute_process(COMMAND ${CMAKE_CXX_COMPILER} --print-file-name=libclang_rt.asan-x86_64.so - OUTPUT_VARIABLE ASAN_LIB_FULL_PATH) - get_filename_component(ASAN_LIB_PATH ${ASAN_LIB_FULL_PATH} DIRECTORY) -else() - set(ASAN_LIB_PATH "$ENV{LD_LIBRARY_PATH}") -endif() - -if(DEFINED RVS_ROCMSMI) - if(NOT RVS_ROCMSMI EQUAL 1) - if(NOT EXISTS "${ROCM_SMI_LIB_DIR}/lib${ROCM_SMI_LIB}.so") - message("ERROR: rocm_smi library can't be found!...") - RETURN() - endif() - endif() -endif() - -## define include directories -include_directories(./ ../ ${ROCM_SMI_INC_DIR} ${YAML_CPP_INCLUDE_DIR}) -# Add directories to look for library files to link -link_directories(${RVS_LIB_DIR} ${ROCM_SMI_LIB_DIR} ${ASAN_LIB_PATH} ${ROCM_SMI_LIB} ${HIPRAND_LIB_DIR} ${ROCRAND_LIB_DIR} ${HIPBLASLT_LIB_DIR}) -## additional libraries -set (PROJECT_LINK_LIBS rvslib libpthread.so libpci.so libm.so) - -## define source files -set(SOURCES src/rvs_module.cpp src/action.cpp src/worker.cpp) - - -## define target -add_library( ${RVS_TARGET} SHARED ${SOURCES}) -set_target_properties(${RVS_TARGET} PROPERTIES - SUFFIX .so.${LIB_VERSION_STRING} - LIBRARY_OUTPUT_DIRECTORY ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}) -target_link_libraries(${RVS_TARGET} ${PROJECT_LINK_LIBS} ${ROCM_SMI_LIB}) -add_dependencies(${RVS_TARGET} rvslib) - -add_custom_command(TARGET ${RVS_TARGET} POST_BUILD -COMMAND ln -fs ./lib${RVS}.so.${LIB_VERSION_STRING} lib${RVS}.so.${VERSION_MAJOR} WORKING_DIRECTORY ${CMAKE_RUNTIME_OUTPUT_DIRECTORY} -COMMAND ln -fs ./lib${RVS}.so.${VERSION_MAJOR} lib${RVS}.so WORKING_DIRECTORY ${CMAKE_RUNTIME_OUTPUT_DIRECTORY} -) - -install(TARGETS ${RVS_TARGET} LIBRARY DESTINATION ${CPACK_PACKAGING_INSTALL_PREFIX}/${CMAKE_INSTALL_LIBDIR}/rvs COMPONENT rvsmodule) -install(FILES "${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/lib${RVS}.so.${VERSION_MAJOR}" - DESTINATION ${CPACK_PACKAGING_INSTALL_PREFIX}/${CMAKE_INSTALL_LIBDIR}/rvs COMPONENT rvsmodule) -install(FILES "${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/lib${RVS}.so" - DESTINATION ${CPACK_PACKAGING_INSTALL_PREFIX}/${CMAKE_INSTALL_LIBDIR}/rvs COMPONENT rvsmodule) - -# TEST SECTION -if (RVS_BUILD_TESTS) - add_custom_command(TARGET ${RVS_TARGET} POST_BUILD - COMMAND ln -fs ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/lib${RVS}.so.${VERSION_MAJOR} ${RVS_BINTEST_FOLDER}/lib${RVS}.so WORKING_DIRECTORY ${CMAKE_RUNTIME_OUTPUT_DIRECTORY} - ) - include(${CMAKE_CURRENT_SOURCE_DIR}/tests.cmake) -endif() diff --git a/gm.so/include/action.h b/gm.so/include/action.h deleted file mode 100644 index a659f18a9..000000000 --- a/gm.so/include/action.h +++ /dev/null @@ -1,84 +0,0 @@ -/******************************************************************************* - * - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ - -#ifndef GM_SO_INCLUDE_ACTION_H_ -#define GM_SO_INCLUDE_ACTION_H_ - -#include -#include - -#include "include/rvsactionbase.h" -#include "include/metrics.h" - -using std::string; - -/** - * @class gm_action - * @ingroup GM - * - * @brief GM action implementation class - * - * Derives from rvs::actionbase and implements actual action functionality - * in its run() method. - * - */ - -class gm_action : public rvs::actionbase { - public: - gm_action(); - virtual ~gm_action(); - - virtual int run(void); - - protected: -/** - * @brief gets the number of ROCm compatible AMD GPUs - * @return run number of GPUs - */ - int get_num_amd_gpu_devices(void); - bool get_all_gm_config_keys(void); - int get_bounds(const char* pMetric); - - protected: - //! true if test has to be aborted on bounds violation - bool prop_terminate; - //! true if forced termination is required - bool prop_force; - //! configuration 'sample_interval'' key - uint64_t sample_interval; - - friend class Worker; - - protected: - //! device_irq and metric bounds - std::map property_bounds; - - private: - //! JSON roor node helper var - void* json_root_node; -}; - -#endif // GM_SO_INCLUDE_ACTION_H_ diff --git a/gm.so/include/metrics.h b/gm.so/include/metrics.h deleted file mode 100644 index b284dc9b9..000000000 --- a/gm.so/include/metrics.h +++ /dev/null @@ -1,88 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2023 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#ifndef GM_SO_INCLUDE_METRICS_H_ -#define GM_SO_INCLUDE_METRICS_H_ - -//! monitored metric and its bound values -typedef struct { - //! true if metric observed - bool mon_metric; - //! true if bounds checked - bool check_bounds; - //! bound max_val - uint32_t max_val; - //! bound min_val - uint32_t min_val; -} Metric_bound; - -//! number of violations for metrics -typedef struct { - //! gpu_id - uint32_t gpu_id; - //! number of temperature violation - int temp_violation; - //! number of clock violation - int clock_violation; - //! number of mem_clock violation - int mem_clock_violation; - //! number of fan violation - int fan_violation; - //! number of power violation - int power_violation; -} Metric_violation; - -//! current metric values -typedef struct { - //! gpu_id - uint32_t gpu_id; - //! current temperature value - int64_t temp; - //! current clock value - uint64_t clock; - //! current mem_clock value - uint64_t mem_clock; - //! current fan percentage - uint32_t fan; - //! current power value - uint32_t power; -} Metric_value; - -//! average metric values -typedef struct { - //! gpu_id - uint32_t gpu_id; - //! average temperature - int64_t av_temp; - //! average clock - uint64_t av_clock; - //! average mem_clock - uint64_t av_mem_clock; - //! average fan - uint64_t av_fan; - //! average power - float av_power; -} Metric_avg; - -#endif // GM_SO_INCLUDE_METRICS_H_ diff --git a/gm.so/include/rvs_module.h b/gm.so/include/rvs_module.h deleted file mode 100644 index 21d02610c..000000000 --- a/gm.so/include/rvs_module.h +++ /dev/null @@ -1,30 +0,0 @@ -/******************************************************************************* * - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#ifndef GM_SO_INCLUDE_RVS_MODULE_H_ -#define GM_SO_INCLUDE_RVS_MODULE_H_ - -#include "include/rvsliblog.h" - -#endif // GM_SO_INCLUDE_RVS_MODULE_H_ diff --git a/gm.so/include/worker.h b/gm.so/include/worker.h deleted file mode 100644 index 7e2b908f5..000000000 --- a/gm.so/include/worker.h +++ /dev/null @@ -1,125 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#ifndef GM_SO_INCLUDE_WORKER_H_ -#define GM_SO_INCLUDE_WORKER_H_ - -#include -#include - -#include "include/rvsthreadbase.h" -#include "include/rvsactionbase.h" -#include "include/metrics.h" -#include "include/action.h" - -/** - * @class Worker - * @ingroup GM - * - * @brief Monitoring implementation class - * - * Derives from rvs::ThreadBase and implements actual monitoring functionality - * in its run() method. - * - */ - -class Worker : public rvs::ThreadBase { - public: - public: - Worker(); - virtual ~Worker(); - - void stop(void); - //! Sets initiating action name - void set_name(const std::string& name) { action_name = name; } - //! sets action - void set_action(const gm_action& _action) { action = _action; } - //! sets stopping action name - void set_stop_name(const std::string& name) { stop_action_name = name; } - //! Sets device indices for filtering - void set_dv_hdl(const std::map& DvHdl) { - dv_hdl = DvHdl; - } - //! Sets JSON flag - void json(const bool flag) { bjson = flag; } - //! Returns initiating action name -// const std::string& get_name(void) { return action_name; } - //! sets sample interval - void set_sample_int(int interval) { sample_interval = interval; } - //! sets log interval - void set_log_int(int interval) { log_interval = interval; } - //! sets terminate key - void set_terminate(bool term_true) { term = term_true; } - //! sets force key - void set_force(bool flag) { force = flag; } - //! sets true/false for metric - void set_metr_mon(std::string metr_name, bool metr_true); - //! sets bound values for metric - void set_bound(const std::map& Bound) { - bounds = Bound; - } - //! gets irq of device - const std::string get_irq(const std::string path); - //! gets power of device - int get_power(const std::string path); - //! prints captured metric values - void do_metric_values(void); - - protected: - virtual void run(void); - - protected: - //! Name of the action which initiated monitoring - std::string action_name; - //! action instance - gm_action action; - //! Name of the action which stops monitoring - std::string stop_action_name; - //! sample interval - int sample_interval; - //! log interval; - int log_interval; - //! terminate key - bool term; - //! force key - bool force; - //! TRUE if JSON output is required - bool bjson; - //! Loops while TRUE - bool brun; - //! list of smi_lib device handles to monitor - std::map dv_hdl; - //! number of times of get metric - int count; - //! dv_hdl and metric bounds - std::map bounds; - //! dv_hdl and metrics violation - std::map met_violation; - //! dv_hdl and current metric values - std::map met_value; - //! dv_hdl and current metric values - std::map met_avg; -}; - -#endif // GM_SO_INCLUDE_WORKER_H_ diff --git a/gm.so/src/action.cpp b/gm.so/src/action.cpp deleted file mode 100644 index 56d9a57c6..000000000 --- a/gm.so/src/action.cpp +++ /dev/null @@ -1,348 +0,0 @@ -/******************************************************************************* - * - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ - -#include "include/action.h" - -#include -#include -#include -#include - -#include "include/rvs_key_def.h" -#include "include/rvsloglp.h" -#include "include/rvs_module.h" -#include "include/rvs_util.h" -#include "include/gpu_util.h" -#include "include/rsmi_util.h" -#include "include/worker.h" - -#define JSON_CREATE_NODE_ERROR "JSON cannot create node" - -#define GM_TEMP "temp" -#define GM_CLOCK "clock" -#define GM_MEM_CLOCK "mem_clock" -#define GM_FAN "fan" -#define GM_POWER "power" -#define GM_FORCE "force" -static constexpr auto MODULE_NAME = "gm"; -static constexpr auto MODULE_NAME_CAPS = "GM"; - -extern Worker* pworker; - -/** - * default class constructor - */ -gm_action::gm_action() { - bjson = false; - json_root_node = nullptr; - module_name = MODULE_NAME; - property_bounds.insert(std::pair - (GM_TEMP, {false, false, 0, 0})); - property_bounds.insert(std::pair - (GM_CLOCK, {false, false, 0, 0})); - property_bounds.insert(std::pair - (GM_MEM_CLOCK, {false, false, 0, 0})); - property_bounds.insert(std::pair - (GM_FAN, {false, false, 0, 0})); - property_bounds.insert(std::pair - (GM_POWER, {false, false, 0, 0})); -} - -/** - * class destructor - */ -gm_action::~gm_action() { - property.clear(); -} - - -/** - * @brief Read configuration 'metric:' key and store it into property_bounds - * array. - * @param pMetric Metric name - * @return 0 - OK - * @return 1 - syntax error - */ -int gm_action::get_bounds(const char* pMetric) { - std::string smetric("metrics."); - smetric += pMetric; - - std::string sval; - if (!has_property(smetric, &sval)) { - return 2; - } - - Metric_bound bound_; - int error; - std::vector values = str_split(sval, YAML_DEVICE_PROP_DELIMITER); - if (values.size() == 3) { - bound_.mon_metric = true; - bound_.check_bounds = (values[0] == "true") ? true : false; - error = rvs_util_parse(values[1], &bound_.max_val); - if (error) { - return 1; - } - error = rvs_util_parse(values[2], &bound_.min_val); - if (error) { - return 1; - } - property_bounds[std::string(pMetric)] = bound_; - } else { - return 1; - } - - return 0; -} - -/** - * @brief reads all GM specific configuration keys from - * the module's properties collection - * @return true if no fatal error occured, false otherwise - */ -bool gm_action::get_all_gm_config_keys(void) { - string msg; - bool sts = true; - - if (get_bounds(GM_TEMP) == 1) { - msg = "Invalid 'metrics." + - std::string(GM_TEMP) + "' key."; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - sts = false; - } - - if (get_bounds(GM_CLOCK) == 1) { - msg = "Invalid 'metrics." + - std::string(GM_CLOCK) + "' key."; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - sts = false; - } - - if (get_bounds(GM_MEM_CLOCK) == 1) { - msg = "Invalid 'metrics." + - std::string(GM_MEM_CLOCK) + "' key."; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - sts = false; - } - - if (get_bounds(GM_FAN) == 1) { - msg = "Invalid 'metrics." + - std::string(GM_FAN) + "' key."; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - sts = false; - } - - if (get_bounds(GM_POWER) == 1) { - msg = "Invalid 'metrics." + - std::string(GM_POWER) + "' key."; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - sts = false; - } - - if (property_get(GM_FORCE, &prop_force, false)) { - msg = "Invalid '" + std::string(GM_FORCE) + "' key."; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - sts = false; - } - - if (property_get(RVS_CONF_TERMINATE_KEY, &prop_terminate, false)) { - msg = "Invalid 'terminate' key."; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - sts = false; - } - if (property_get_int(RVS_CONF_SAMPLE_INTERVAL_KEY, - &sample_interval, 500u)) { - msg = "Invalid '" +std::string(RVS_CONF_SAMPLE_INTERVAL_KEY) + "' key."; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - sts = false; - } - if (property_log_interval < sample_interval) { - msg = "Log interval has the lower value than the sample interval."; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - sts = false; - } - - return sts; -} -/** - * @brief Implements action functionality - * - * Functionality: - * - * @return 0 - success. non-zero otherwise - * - * */ -int gm_action::run(void) { - string msg; - amdsmi_status_t status; - rvs::action_result_t action_result; - - // if monitoring is already running, stop it - // (it will be restarted if needed) - RVSTRACE_ - if (pworker) { - RVSTRACE_ - // (give thread chance to start) - sleep(2); - pworker->set_stop_name(property["name"]); - pworker->stop(); - delete pworker; - pworker = nullptr; - } - // this action should stop monitoring? - if (property["monitor"] != "true") { - RVSTRACE_ - // already done, just return - action_result.state = rvs::actionstate::ACTION_COMPLETED; - action_result.status = rvs::actionstatus::ACTION_FAILED; - action_result.output = "GM Module action " + action_name + " completed"; - action_callback(&action_result); - return 0; - } - - RVSTRACE_ - // start new monitoring - if (!get_all_common_config_keys()) { - RVSTRACE_ - action_result.state = rvs::actionstate::ACTION_COMPLETED; - action_result.status = rvs::actionstatus::ACTION_FAILED; - action_result.output = "Error in common configuration keys."; - action_callback(&action_result); - return -1; - } - - if (!get_all_gm_config_keys()) { - RVSTRACE_ - - action_result.state = rvs::actionstate::ACTION_COMPLETED; - action_result.status = rvs::actionstatus::ACTION_FAILED; - action_result.output = "Error in GM configuration keys."; - action_callback(&action_result); - return -1; - } - - RVSTRACE_ - - // if 'device: all' get all AMD GPU IDs - if (property_device_all) { - gpu_get_all_gpu_id(&property_device); - } - - // apply device_id filtering if needed - if (property_device_id > 0) { - RVSTRACE_ - std::vector gpu_id_filtered; - for (auto it = property_device.begin(); it != property_device.end(); it++) { - RVSTRACE_ - - uint16_t _dev_id; - if (rvs::gpulist::gpu2device(*it, &_dev_id)) { - RVSTRACE_ - // if not found just continue - continue; - } - - if (_dev_id == property_device_id) { - RVSTRACE_ - gpu_id_filtered.push_back(*it); - } - } - property_device = gpu_id_filtered; - } - - RVSTRACE_ - - // verify that the resulting array is not empty - if (property_device.size() < 1) { - msg = "No devices match filtering criteria."; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - - action_result.state = rvs::actionstate::ACTION_COMPLETED; - action_result.status = rvs::actionstatus::ACTION_FAILED; - action_result.output = msg; - action_callback(&action_result); - return -1; - } - - // convert GPU ID into amd_smi_lib device hdl - std::map dv_hdl; - for (auto it = property_device.begin(); it != property_device.end(); it++) { - RVSTRACE_ - uint16_t location_id; - if (rvs::gpulist::gpu2location(*it, &location_id)) { - msg = "Could not obtain BDF for GPU ID: "; - msg += std::to_string(*it); - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - - action_result.state = rvs::actionstate::ACTION_COMPLETED; - action_result.status = rvs::actionstatus::ACTION_FAILED; - action_result.output = msg; - action_callback(&action_result); - return -1; - } - amdsmi_processor_handle ix; - status = rvs::smi_dev_ind_get(location_id, &ix); - if(status == AMDSMI_STATUS_SUCCESS) { - dv_hdl.insert(std::pair(*it, ix)); - } - } - - pworker = new Worker(); - pworker->set_name(action_name); - pworker->set_action(*this); - pworker->json(bjson); - pworker->set_sample_int(sample_interval); - pworker->set_log_int(property_log_interval); - pworker->set_terminate(prop_terminate); - if (prop_force) - pworker->set_force(true); - - // set stop name before start - pworker->set_stop_name(action_name); - // set array of device indices to monitor - pworker->set_dv_hdl(dv_hdl); - // set bounds map - pworker->set_bound(property_bounds); - - RVSTRACE_ - // start worker thread - pworker->start(); - - // this should be used only for testing purposes - if (property_duration) { - RVSTRACE_ - sleep(property_duration); - } - - RVSTRACE_ - - action_result.state = rvs::actionstate::ACTION_COMPLETED; - action_result.status = rvs::actionstatus::ACTION_SUCCESS; - action_result.output = "GM Module action " + action_name + " completed"; - action_callback(&action_result); - - return 0; -} - diff --git a/gm.so/src/rvs_module.cpp b/gm.so/src/rvs_module.cpp deleted file mode 100644 index 7ea8d1a98..000000000 --- a/gm.so/src/rvs_module.cpp +++ /dev/null @@ -1,119 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#include "include/rvs_module.h" - -#include -#include - -#include "amd_smi/amdsmi.h" - -#include "include/action.h" -#include "include/rvsloglp.h" -#include "include/worker.h" -#include "include/gpu_util.h" - -/** - * @defgroup GM GM Module - * - * @brief GPU Monitor module - * - * The GPU monitor tool is capable of running on one, some or all of the GPU(s) - * installed and will - * report various information at regular intervals. The module can be configured - * to halt another - * RVS modules execution if one of the quantities exceeds a specified boundary - * value. - */ - -Worker* pworker; - -extern "C" int rvs_module_has_interface(int iid) { - int sts = 0; - switch (iid) { - case 0: - case 1: - sts = 1; - } - return sts; -} - -extern "C" const char* rvs_module_get_description(void) { - return "The GPU monitor tool is capable of running on one, some or all of the GPU(s) installed and will report various information \n\tat regular intervals."; -} - -extern "C" const char* rvs_module_get_config(void) { - return "monitor (bool)"; -} - -extern "C" const char* rvs_module_get_output(void) { - return "state (string)"; -} - -extern "C" int rvs_module_init(void* pMi) { - rvs::lp::Initialize(static_cast(pMi)); - RVSTRACE_ - rvs::gpulist::Initialize(); - return 0; -} - -extern "C" int rvs_module_terminate(void) { - RVSTRACE_ - if (pworker) { - RVSTRACE_ - pworker->set_stop_name("module_terminate"); - pworker->stop(); - delete pworker; - pworker = nullptr; - } - RVSTRACE_ - amdsmi_shut_down(); - - return 0; -} - -extern "C" void* rvs_module_action_create(void) { - return static_cast(new gm_action); -} - -extern "C" int rvs_module_action_destroy(void* pAction) { - delete static_cast(pAction); - return 0; -} - -extern "C" int rvs_module_action_property_set(void* pAction, const char* Key, - const char* Val) { - return static_cast(pAction)->property_set(Key, Val); -} - -extern "C" int rvs_module_action_callback_set(void* pAction, - rvs::callback_t callback, - void * user_param) { - return static_cast(pAction)->callback_set(callback, user_param); -} - -extern "C" int rvs_module_action_run(void* pAction) { - return static_cast(pAction)->run(); -} - diff --git a/gm.so/src/worker.cpp b/gm.so/src/worker.cpp deleted file mode 100644 index 19ace89da..000000000 --- a/gm.so/src/worker.cpp +++ /dev/null @@ -1,627 +0,0 @@ -/******************************************************************************* -* - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy -of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to -do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in -all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - -*******************************************************************************/ -#include "include/worker.h" - -#include -#include -#include -#include - -#include "include/rvs_module.h" -#include "include/gpu_util.h" -#include "include/rvs_util.h" -#include "include/rvsloglp.h" -#include "include/rvstimer.h" -#include "include/rsmi_util.h" - -#define MODULE_NAME_CAPS "GM" - -#define PCI_ALLOC_ERROR "pci_alloc() error" -#define GM_RESULT_FAIL_MESSAGE "FALSE" -#define IRQ_PATH_MAX_LENGTH 256 -#define MODULE_NAME "gm" -#define GM_TEMP "temp" -#define GM_CLOCK "clock" -#define GM_MEM_CLOCK "mem_clock" -#define GM_FAN "fan" -#define GM_POWER "power" - - -// collection of allowed metrics -const char* metric_names[] = - { GM_TEMP, GM_CLOCK, GM_MEM_CLOCK, GM_FAN, GM_POWER - }; - - -Worker::Worker() { - force = false; -} -Worker::~Worker() {} - -/** - * @brief Prints current metric values at every log_interval msec. - */ -void Worker::do_metric_values() { - std::string msg; - unsigned int sec; - unsigned int usec; - void* r; - - // get timestamp - rvs::lp::get_ticks(&sec, &usec); - // add JSON output - r = rvs::lp::LogRecordCreate("gm", action_name.c_str(), rvs::loginfo, - sec, usec); - - for (auto it = met_avg.begin(); it != - met_avg.end(); it++) { - if (bounds[GM_TEMP].mon_metric) { - msg = "[" + action_name + "] gm " + - std::to_string((it->second).gpu_id) + " " + GM_TEMP + - " " + std::to_string(met_value[it->first].temp) + "C"; - rvs::lp::Log(msg, rvs::loginfo, sec, usec); - rvs::lp::AddString(r, "info ", msg); - } - if (bounds[GM_CLOCK].mon_metric) { - msg = "[" + action_name + "] gm " + - std::to_string((it->second).gpu_id) + " " + GM_CLOCK + - " " + std::to_string(met_value[it->first].clock) + "MHz"; - rvs::lp::Log(msg, rvs::loginfo, sec, usec); - rvs::lp::AddString(r, "info ", msg); - } - if (bounds[GM_MEM_CLOCK].mon_metric) { - msg = "[" + action_name + "] gm " + - std::to_string((it->second).gpu_id) + " " + GM_MEM_CLOCK + - " " + std::to_string(met_value[it->first].mem_clock) + "MHz"; - rvs::lp::Log(msg, rvs::loginfo, sec, usec); - rvs::lp::AddString(r, "info ", msg); - } - if (bounds[GM_FAN].mon_metric) { - msg = "[" + action_name + "] gm " + - std::to_string((it->second).gpu_id) + " " + GM_FAN + - " " + std::to_string(met_value[it->first].fan) + "%"; - rvs::lp::Log(msg, rvs::loginfo, sec, usec); - rvs::lp::AddString(r, "info ", msg); - } - if (bounds[GM_POWER].mon_metric) { - msg = "[" + action_name + "] gm " + - std::to_string((it->second).gpu_id) + " " + GM_POWER + - " " + std::to_string(static_cast(met_value[it->first].power) / - 1e6) + "Watts"; - rvs::lp::Log(msg, rvs::loginfo, sec, usec); - rvs::lp::AddString(r, "info ", msg); - } - } - rvs::lp::LogRecordFlush(r); -} - -/** - * @brief Thread function - * - * Loops while brun == TRUE and performs polled monitoring avery 1msec. - * - * */ -void Worker::run() { - brun = true; -// std::string val_str; -// std::vector val_vec; - - std::string msg; - amdsmi_status_t status; - amdsmi_frequencies_t f; - uint32_t sensor_ind = 0; - int64_t temperature; - int64_t speed; - uint64_t max_speed; - uint32_t fan_percentage; - uint64_t power; - - unsigned int sec; - unsigned int usec; - void* r; - rvs::action_result_t action_result; - - rvs::timer timer_running(&Worker::do_metric_values, this); - - // get timestamp - rvs::lp::get_ticks(&sec, &usec); - - // add JSON output - r = rvs::lp::LogRecordCreate("gm", action_name.c_str(), rvs::loginfo, - sec, usec); - - // iterate over devices - uint16_t loop_idx=0; - std::map idx_smi_map; - for (auto it = dv_hdl.begin(); it != dv_hdl.end(); it++, loop_idx++) { - RVSTRACE_ - // fill in the info - idx_smi_map[loop_idx] = it->second; - met_avg.insert(std::pair - (loop_idx, {it->first, 0, 0, 0, 0, 0})); - met_violation.insert(std::pair - (loop_idx, {it->first, 0, 0, 0, 0, 0})); - met_value.insert(std::pair - (loop_idx, {it->first, 0, 0, 0, 0, 0})); - - msg = "[" + action_name + "] gm " + std::to_string(it->first) + - " started"; - rvs::lp::Log(msg, rvs::logresults, sec, usec); - rvs::lp::AddString(r, "device", std::to_string(loop_idx)); - for (auto itb = bounds.begin(); itb != bounds.end(); itb++) { - RVSTRACE_ - - if (itb->second.mon_metric) { - msg = "[" + action_name + "] " + MODULE_NAME + " " + - std::to_string(it->first) + " " + "monitoring " + - itb->first; - if (itb->second.check_bounds) { - msg+= " bounds min: " + std::to_string(itb->second.min_val) + - " max: " + std::to_string(itb->second.max_val); - } - rvs::lp::Log(msg, rvs::loginfo); - rvs::lp::AddString(r, itb->first, msg); - } - } - } - - rvs::lp::LogRecordFlush(r); - // if log_interval timer starts - if (log_interval) { - timer_running.start(log_interval); - } - - count = 0; - - // worker thread has started - while (brun) { - RVSTRACE_ - uint16_t loop_idx=0; - for (auto it = dv_hdl.begin(); it != dv_hdl.end(); it++, loop_idx++) { - - auto ix = it->second; - int32_t gpuid = it->first; - RVSTRACE_ - if (bounds[GM_MEM_CLOCK].mon_metric) { - RVSTRACE_ - status = amdsmi_get_clk_freq(ix, AMDSMI_CLK_TYPE_MEM, &f); - uint64_t mhz = f.frequency[f.current]/1e6; - met_value[loop_idx].mem_clock = mhz; - if (!(mhz >= bounds[GM_MEM_CLOCK].min_val && mhz <= - bounds[GM_MEM_CLOCK].max_val) && - bounds[GM_MEM_CLOCK].check_bounds) { - RVSTRACE_ - // write info and increase number of violations - msg = "[" + action_name + "] " + MODULE_NAME + " " + - std::to_string(gpuid) + " " + - GM_MEM_CLOCK + " " + "bounds violation " + - std::to_string(mhz) + "MHz"; - rvs::lp::Log(msg, rvs::loginfo); - - action_result.state = rvs::actionstate::ACTION_RUNNING; - action_result.status = rvs::actionstatus::ACTION_SUCCESS; - action_result.output = msg.c_str(); - action.action_callback(&action_result); - - met_violation[loop_idx].mem_clock_violation++; - if (term) { - RVSTRACE_ - if (force) { - RVSTRACE_ - // stop logging - rvs::lp::Stop(1); - // force exit - exit(EXIT_FAILURE); - } else { - RVSTRACE_ - // just signal stop processing - rvs::lp::Stop(0); - } - brun = false; - } - RVSTRACE_ - } - RVSTRACE_ - met_avg[loop_idx].av_mem_clock += mhz; - } - RVSTRACE_ - - if (bounds[GM_CLOCK].mon_metric) { - RVSTRACE_ - status = amdsmi_get_clk_freq(ix, - AMDSMI_CLK_TYPE_SYS , &f); - uint32_t mhz = f.frequency[f.current]/1e6; - met_value[loop_idx].clock = mhz; - if (!(mhz >= bounds[GM_CLOCK].min_val && mhz <= - bounds[GM_CLOCK].max_val) && - bounds[GM_CLOCK].check_bounds) { - RVSTRACE_ - // write info - msg = "[" + action_name + "] " + MODULE_NAME + " " + - std::to_string(met_avg[loop_idx].gpu_id) + " " + - GM_CLOCK + " " + "bounds violation " + - std::to_string(mhz) + "MHz"; - rvs::lp::Log(msg, rvs::loginfo); - - action_result.state = rvs::actionstate::ACTION_RUNNING; - action_result.status = rvs::actionstatus::ACTION_SUCCESS; - action_result.output = msg.c_str(); - action.action_callback(&action_result); - - met_violation[loop_idx].clock_violation++; - if (term) { - RVSTRACE_ - if (force) { - RVSTRACE_ - // stop logging - rvs::lp::Stop(1); - // force exit - exit(EXIT_FAILURE); - } else { - RVSTRACE_ - // just signal stop processing - rvs::lp::Stop(0); - } - RVSTRACE_ - brun = false; - } - RVSTRACE_ - } - met_avg[loop_idx].av_clock += mhz; - RVSTRACE_ - } - - RVSTRACE_ - if (bounds[GM_TEMP].mon_metric) { - RVSTRACE_ - - // Get GPU's current junction temperature - status = amdsmi_get_temp_metric(ix, AMDSMI_TEMPERATURE_TYPE_JUNCTION , - AMDSMI_TEMP_CURRENT , &temperature); - -#ifdef UT_TCD_1 - status = AMDSMI_STATUS_UNKNOWN_ERROR; -#endif // UT_TCD_1 - if (status == AMDSMI_STATUS_SUCCESS) { - RVSTRACE_ - int64_t temper = temperature/1000; - met_value[loop_idx].temp = temper; - met_avg[loop_idx].av_temp += temper; - if (!(temper >= bounds[GM_TEMP].min_val && temper <= - bounds[GM_TEMP].max_val) && - bounds[GM_TEMP].check_bounds) { - RVSTRACE_ - // write info - msg = "[" + action_name + "] " + MODULE_NAME + " " + - std::to_string(met_avg[loop_idx].gpu_id) + " " + - + GM_TEMP + " " + "bounds violation " + - std::to_string(temper) + "C"; - rvs::lp::Log(msg, rvs::loginfo); - - - action_result.state = rvs::actionstate::ACTION_RUNNING; - action_result.status = rvs::actionstatus::ACTION_SUCCESS; - action_result.output = msg.c_str(); - action.action_callback(&action_result); - - met_violation[loop_idx].temp_violation++; - if (term) { - RVSTRACE_ - if (force) { - RVSTRACE_ - // stop logging - rvs::lp::Stop(1); - // force exit - RVSTRACE_ - exit(EXIT_FAILURE); - } else { - RVSTRACE_ - // just signal stop processing - rvs::lp::Stop(0); - } - brun = false; - RVSTRACE_ - } - RVSTRACE_ - } - RVSTRACE_ - } else { - RVSTRACE_ - msg = "[" + action_name + "] " + MODULE_NAME + " " + - std::to_string(met_avg[loop_idx].gpu_id) + " " + - GM_TEMP + " Not available"; - rvs::lp::Log(msg, rvs::loginfo); - } - RVSTRACE_ - } - - RVSTRACE_ - if (bounds[GM_FAN].mon_metric) { - RVSTRACE_ - - speed = 0; - max_speed = 0; - status = amdsmi_get_gpu_fan_speed(ix, sensor_ind, &speed); - if (status == AMDSMI_STATUS_SUCCESS ) { - status = amdsmi_get_gpu_fan_speed_max(ix, sensor_ind, &max_speed); - } - - /* Calculate fan speed percentage */ - if ((speed != 0) && (max_speed != 0)) { - fan_percentage = static_cast((speed * 100)/max_speed); - } - -#ifdef UT_TCD_1 - status = AMDSMI_STATUS_UNKNOWN_ERROR ; -#endif // UT_TCD_1 - if (status == AMDSMI_STATUS_SUCCESS) { - RVSTRACE_ - met_value[loop_idx].fan = fan_percentage; - met_avg[loop_idx].av_fan += static_cast (fan_percentage); - - if (!(fan_percentage >= bounds[GM_FAN].min_val && fan_percentage <= - bounds[GM_FAN].max_val) && - bounds[GM_FAN].check_bounds) { - RVSTRACE_ - // write info - msg = "[" + action_name + "] " + MODULE_NAME + " " + - std::to_string(met_avg[loop_idx].gpu_id) + " " + - + GM_FAN + " " + "bounds violation " + - std::to_string(fan_percentage) + "%"; - rvs::lp::Log(msg, rvs::loginfo); - - action_result.state = rvs::actionstate::ACTION_RUNNING; - action_result.status = rvs::actionstatus::ACTION_SUCCESS; - action_result.output = msg.c_str(); - action.action_callback(&action_result); - - met_violation[loop_idx].fan_violation++; - if (term) { - RVSTRACE_ - if (force) { - RVSTRACE_ - // stop logging - rvs::lp::Stop(1); - // force exit - exit(EXIT_FAILURE); - } else { - RVSTRACE_ - // just signal stop processing - rvs::lp::Stop(0); - } - brun = false; - RVSTRACE_ - break; - } - RVSTRACE_ - } - RVSTRACE_ - } else { - RVSTRACE_ - msg = "[" + action_name + "] " + MODULE_NAME + " " + - std::to_string(met_avg[loop_idx].gpu_id) + " " + - GM_FAN + " Not available"; - rvs::lp::Log(msg, rvs::loginfo); - } - RVSTRACE_ - } - - RVSTRACE_ - if (bounds[GM_POWER].mon_metric) { - RVSTRACE_ - amdsmi_power_info_t pwr_info; - status = amdsmi_get_power_info(ix, &pwr_info); - met_value[loop_idx].power = pwr_info.socket_power; - met_avg[loop_idx].av_power += pwr_info.socket_power; - if (bounds[GM_POWER].check_bounds) { - RVSTRACE_ - if (power < bounds[GM_POWER].min_val * 1000000 || - power > bounds[GM_POWER].max_val * 1000000) { - RVSTRACE_ - // write info - msg = "[" + action_name + "] " + MODULE_NAME + " " + - std::to_string(met_avg[loop_idx].gpu_id) + " " + - GM_POWER + " " + "bounds violation " + - std::to_string(static_cast(power / 1e6)) + "Watts"; - rvs::lp::Log(msg, rvs::loginfo); - - action_result.state = rvs::actionstate::ACTION_RUNNING; - action_result.status = rvs::actionstatus::ACTION_SUCCESS; - action_result.output = msg.c_str(); - action.action_callback(&action_result); - - met_violation[loop_idx].power_violation++; - if (term) { - RVSTRACE_ - if (force) { - RVSTRACE_ - // stop logging - rvs::lp::Stop(1); - // force exit - exit(EXIT_FAILURE); - } else { - RVSTRACE_ - // just signal stop processing - rvs::lp::Stop(0); - } - brun = false; - RVSTRACE_ - } - RVSTRACE_ - } - RVSTRACE_ - } - RVSTRACE_ - } - RVSTRACE_ - } - count++; - sleep(sample_interval); - RVSTRACE_ - } - - RVSTRACE_ - timer_running.stop(); - sleep(200); - - // get timestamp - rvs::lp::get_ticks(&sec, &usec); - - for (auto it = met_avg.begin(); - it != met_avg.end(); it++) { - RVSTRACE_ - // add std::string output - msg = "[" + action_name + "] gm " + - std::to_string((it->second).gpu_id) + " stopped"; - rvs::lp::Log(msg, rvs::logresults, sec, usec); - } - - RVSTRACE_ -} - - -/** - * @brief Stops monitoring - * - * Sets brun member to FALSE thus signaling end of monitoring. - * Then it waits for std::thread to exit before returning. - * - * */ -void Worker::stop() { - RVSTRACE_ - rvs::lp::Log("[" + stop_action_name + "] gm in Worker::stop()", - rvs::logtrace); - std::string msg; - unsigned int sec; - unsigned int usec; - void* r; - // get timestamp - rvs::lp::get_ticks(&sec, &usec); - // add JSON output - r = rvs::lp::LogRecordCreate("result", action_name.c_str(), rvs::logresults, - sec, usec); - // reset "run" flag - brun = false; - // (give thread chance to finish processing and exit) - sleep(200); - - if (count != 0) { - RVSTRACE_ - for (auto it = met_avg.begin(); it != - met_avg.end(); it++) { - RVSTRACE_ - if (bounds[GM_TEMP].mon_metric) { - RVSTRACE_ - msg = "[" + action_name + "] gm " + - std::to_string((it->second).gpu_id) + " " + - GM_TEMP + " violations " + - std::to_string(met_violation[it->first].temp_violation); - rvs::lp::Log(msg, rvs::logresults, sec, usec); - rvs::lp::AddString(r, "result", msg); - msg = "[" + action_name + "] gm " + - std::to_string((it->second).gpu_id) + " "+ GM_TEMP + " average " + - std::to_string((it->second).av_temp/count) + "C"; - rvs::lp::Log(msg, rvs::logresults, sec, usec); - rvs::lp::AddString(r, "result", msg); - } - RVSTRACE_ - if (bounds[GM_CLOCK].mon_metric) { - RVSTRACE_ - msg = "[" + action_name + "] gm " + - std::to_string((it->second).gpu_id) + " " + - GM_CLOCK + " violations " + - std::to_string(met_violation[it->first].clock_violation); - rvs::lp::Log(msg, rvs::logresults, sec, usec); - rvs::lp::AddString(r, "result", msg); - msg = "[" + action_name + "] gm " + - std::to_string((it->second).gpu_id) + " " + GM_CLOCK + " average " + - std::to_string((it->second).av_clock/count) + "MHz"; - rvs::lp::Log(msg, rvs::logresults, sec, usec); - rvs::lp::AddString(r, "result", msg); - } - RVSTRACE_ - if (bounds[GM_MEM_CLOCK].mon_metric) { - RVSTRACE_ - msg = "[" + action_name + "] gm " + - std::to_string((it->second).gpu_id) + - " " + GM_MEM_CLOCK + " violations " + - std::to_string(met_violation[it->first].mem_clock_violation); - rvs::lp::Log(msg, rvs::logresults, sec, usec); - rvs::lp::AddString(r, "result", msg); - msg = "[" + action_name + "] gm " + - std::to_string((it->second).gpu_id) + " " + - GM_MEM_CLOCK + " average " + - std::to_string((it->second).av_mem_clock/count) + "MHz"; - rvs::lp::Log(msg, rvs::logresults, sec, usec); - rvs::lp::AddString(r, "result", msg); - } - RVSTRACE_ - if (bounds[GM_FAN].mon_metric) { - RVSTRACE_ - msg = "[" + action_name + "] gm " + - std::to_string((it->second).gpu_id) + " " + GM_FAN +" violations " + - std::to_string(met_violation[it->first].fan_violation); - rvs::lp::Log(msg, rvs::logresults, sec, usec); - msg = "[" + action_name + "] gm " + - std::to_string((it->second).gpu_id) + " " + GM_FAN + " average " + - std::to_string(static_cast(((it->second).av_fan)/count)) + "%"; - rvs::lp::Log(msg, rvs::logresults, sec, usec); - } - RVSTRACE_ - if (bounds[GM_POWER].mon_metric) { - RVSTRACE_ - msg = "[" + action_name + "] gm " + - std::to_string((it->second).gpu_id) + " " + - GM_POWER + " violations " + - std::to_string(met_violation[it->first].power_violation); - rvs::lp::Log(msg, rvs::logresults, sec, usec); - rvs::lp::AddString(r, "result", msg); - msg = "[" + action_name + "] gm " + - std::to_string((it->second).gpu_id) + " " + GM_POWER + " average " + - std::to_string(static_cast(((it->second).av_power) / - count/1e6)) + "Watts"; - rvs::lp::Log(msg, rvs::logresults, sec, usec); - rvs::lp::AddString(r, "result", msg); - } - RVSTRACE_ - } - RVSTRACE_ - } - RVSTRACE_ - rvs::lp::LogRecordFlush(r); - - // wait a bit to make sure thread has exited - try { - if (t.joinable()) - t.join(); - } - catch(...) { - } -} diff --git a/gm.so/test/test_1.cpp b/gm.so/test/test_1.cpp deleted file mode 100644 index ff9c47566..000000000 --- a/gm.so/test/test_1.cpp +++ /dev/null @@ -1,52 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#include "gtest/gtest.h" -#include "include/action.h" -#include "include/worker.h" -#include "include/gpu_util.h" -#include "amd_smi/amdsmi.h" -Worker* pworker; - -TEST(gm, coverage_rsmi_failure) { - amdsmi_init(AMDSMI_INIT_AMD_GPUS); - rvs::gpulist::Initialize(); - pworker = nullptr; - gm_action* pa = new gm_action; - ASSERT_NE(pa, nullptr); - pa->property_set("monitor", "true"); - pa->property_set("name", "unit_test"); - pa->property_set("device", "all"); - pa->property_set("terminate", "true"); - pa->property_set("metrics.temp", "true 30 0"); - pa->property_set("metrics.fan", "true 100 0"); - pa->property_set("metrics.clock", "true 1500 0"); - pa->property_set("metrics.mem_clock", "true 1500 0"); - pa->property_set("duration", "1000"); - pa->run(); - delete pa; - pworker->stop(); - delete pworker; - amdsmi_shut_down(); -} diff --git a/gm.so/tests.cmake b/gm.so/tests.cmake deleted file mode 100644 index 8cfc648d2..000000000 --- a/gm.so/tests.cmake +++ /dev/null @@ -1,54 +0,0 @@ -################################################################################ -## -## Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. -## -## MIT LICENSE: -## Permission is hereby granted, free of charge, to any person obtaining a copy of -## this software and associated documentation files (the "Software"), to deal in -## the Software without restriction, including without limitation the rights to -## use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies -## of the Software, and to permit persons to whom the Software is furnished to do -## so, subject to the following conditions: -## -## The above copyright notice and this permission notice shall be included in all -## copies or substantial portions of the Software. -## -## THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -## IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -## FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -## AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -## LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -## OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE -## SOFTWARE. -## -################################################################################ - -set(ROCBLAS_LIB "rocblas") -set(HIPRAND_LIB "hiprand") -set(HIPBLASLT_LIB "hipblaslt") -set(CORE_RUNTIME_NAME "hsa-runtime") -set(CORE_RUNTIME_TARGET "${CORE_RUNTIME_NAME}64") - -find_package(OpenMP) - -set(UT_LINK_LIBS libpthread.so libpci.so libm.so libdl.so ${AMD_SMI_LIB} OpenMP::OpenMP_CXX - ${ROCBLAS_LIB} ${ROC_THUNK_NAME} ${CORE_RUNTIME_TARGET} ${ROCM_CORE} ${YAML_CPP_LIBRARIES} ${HIPRAND_LIB} ${HIPBLASLT_LIB} -) - -# Add directories to look for library files to link -link_directories(${ROCM_SMI_LIB_DIR} ${ROCBLAS_LIB_DIR} ${HIPRAND_LIB_DIR} ${HSA_LIB_DIR} ${HIPBLASLT_LIB_DIR} ${YAML_CPP_LIBRARY_DIR}) - -set (UT_SOURCES src/action.cpp src/worker.cpp -) - -#define additional target compile definitions for tests (if any) -set(tcd.unit.gm.1 UT_TCD_1) - -# add unit tests -include(tests_unit) - -if(RVS_ROCMSMI EQUAL 1) - add_dependencies(unit.gm.1 rvs_rsmi_target) -endif() - -include(tests_conf_logging) diff --git a/gpup.so/CMakeLists.txt b/gpup.so/CMakeLists.txt index d2775b6f9..f1fc9df24 100644 --- a/gpup.so/CMakeLists.txt +++ b/gpup.so/CMakeLists.txt @@ -112,7 +112,7 @@ include_directories(./ ../ include ../include ${YAML_CPP_INCLUDE_DIR}) # Add directories to look for library files to link link_directories(${RVS_LIB_DIR} ${ASAN_LIB_PATH} ${AMD_SMI_LIB_DIR} ${HIPRAND_LIB_DIR} ${ROCRAND_LIB_DIR} ${ROCR_LIB_DIR}) ## additional libraries -set (PROJECT_LINK_LIBS rvslib libpci.so libm.so) +set (PROJECT_LINK_LIBS rvslib libm.so) ## define source files set(SOURCES src/rvs_module.cpp src/action.cpp) diff --git a/gpup.so/src/action.cpp b/gpup.so/src/action.cpp index d2a7d8f97..00178e1c2 100644 --- a/gpup.so/src/action.cpp +++ b/gpup.so/src/action.cpp @@ -514,7 +514,7 @@ int gpup_action::run(void) { } if (!b_gpu_found) { msg = "No device matches criteria from configuration. "; - rvs::lp::Err(msg, MODULE_NAME, action_name); +// rvs::lp::Err(msg, MODULE_NAME, action_name); // Action callback action_result.state = rvs::actionstate::ACTION_COMPLETED; diff --git a/gpup.so/src/rvs_module.cpp b/gpup.so/src/rvs_module.cpp index 42b2a6ba9..0d995f6f2 100644 --- a/gpup.so/src/rvs_module.cpp +++ b/gpup.so/src/rvs_module.cpp @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2025 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -68,7 +68,6 @@ extern "C" int rvs_module_init(void* pMi) { } extern "C" int rvs_module_terminate(void) { - cleanup_logs(); amdsmi_shut_down(); return 0; } diff --git a/gpup.so/tests.cmake b/gpup.so/tests.cmake index 3204b87f0..f24797012 100644 --- a/gpup.so/tests.cmake +++ b/gpup.so/tests.cmake @@ -29,9 +29,7 @@ set(HIPBLASLT_LIB "hipblaslt") set(CORE_RUNTIME_NAME "hsa-runtime") set(CORE_RUNTIME_TARGET "${CORE_RUNTIME_NAME}64") -find_package(OpenMP) - -set(UT_LINK_LIBS libpthread.so libm.so libdl.so ${ROCM_SMI_LIB} OpenMP::OpenMP_CXX +set(UT_LINK_LIBS libpthread.so libm.so libdl.so ${ROCM_SMI_LIB} -fopenmp ${ROCBLAS_LIB} ${ROC_THUNK_NAME} ${CORE_RUNTIME_TARGET} ${ROCM_CORE} ${YAML_CPP_LIBRARIES} ${HIPRAND_LIB} ${HIPBLASLT_LIB}) # Add directories to look for library files to link diff --git a/gst.so/CMakeLists.txt b/gst.so/CMakeLists.txt index bdcb4b08a..7a553ae99 100644 --- a/gst.so/CMakeLists.txt +++ b/gst.so/CMakeLists.txt @@ -146,7 +146,7 @@ include_directories(./ ../ ${ROCR_INC_DIR} ${ROCBLAS_INC_DIR} ${HIP_INC_DIR} ${H # Add directories to look for library files to link link_directories(${RVS_LIB_DIR} ${ROCR_LIB_DIR} ${HIP_LIB_DIR} ${ROCBLAS_LIB_DIR} ${ASAN_LIB_PATH} ${AMD_SMI_LIB_DIR} ${HIPRAND_DIR} ${ROCRAND_DIR}) ## additional libraries -set (PROJECT_LINK_LIBS rvslib libpthread.so libpci.so libm.so) +set (PROJECT_LINK_LIBS rvslib libpthread.so libm.so) ## define source files set(SOURCES src/rvs_module.cpp src/action.cpp src/gst_worker.cpp) diff --git a/gst.so/include/action.h b/gst.so/include/action.h index 9e1373517..2e7b60d2d 100644 --- a/gst.so/include/action.h +++ b/gst.so/include/action.h @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2025 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -25,13 +25,6 @@ #ifndef GST_SO_INCLUDE_ACTION_H_ #define GST_SO_INCLUDE_ACTION_H_ -#ifdef __cplusplus -extern "C" { -#endif -#include -#ifdef __cplusplus -} -#endif #include #include @@ -90,6 +83,8 @@ class gst_action: public rvs::actionbase { //Parameter to heat up uint64_t gst_hot_calls; + //Parameter for warm-up calls before ramp + uint64_t gst_warm_calls; //Tranpose set to none or enabled int gst_trans_a; diff --git a/gst.so/include/gst_worker.h b/gst.so/include/gst_worker.h index 96a4e18f2..8e614bef7 100644 --- a/gst.so/include/gst_worker.h +++ b/gst.so/include/gst_worker.h @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2025 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -125,11 +125,21 @@ class GSTWorker : public rvs::ThreadBase { gst_hot_calls = _hot_calls; } - //! sets hot calls + //! gets hot calls uint64_t get_gst_hot_calls(void) { return gst_hot_calls; } + //! sets warm calls + void set_gst_warm_calls(uint64_t _warm_calls) { + gst_warm_calls = _warm_calls; + } + + //! gets warm calls + uint64_t get_gst_warm_calls(void) { + return gst_warm_calls; + } + //! sets the matrix size void set_matrix_size_a(uint64_t _matrix_size_a) { matrix_size_a = _matrix_size_a; @@ -293,6 +303,8 @@ class GSTWorker : public rvs::ThreadBase { void set_gst_scale_b(std::string _scale_b) { gst_scale_b = _scale_b; } //! set rotating buffer size void set_gst_rotating(uint32_t _rotating) { gst_rotating = _rotating; } + //! get worker job result + bool get_result(void) { return result; } protected: void setup_blas(int *error, std::string *err_description); @@ -353,6 +365,8 @@ class GSTWorker : public rvs::ThreadBase { std::string matrix_init; //num of hot calls uint64_t gst_hot_calls; + //num of warm-up calls during ramp period + uint64_t gst_warm_calls; //! actual ramp time in case the GPU achieves the given target_stress Gflops uint64_t ramp_actual_time; //! rvs_blas pointer @@ -411,6 +425,8 @@ class GSTWorker : public rvs::ThreadBase { std::string gst_scale_b; //! Rotating buffer size uint32_t gst_rotating; + //! Worker job result + bool result; }; #endif // GST_SO_INCLUDE_GST_WORKER_H_ diff --git a/gst.so/src/action.cpp b/gst.so/src/action.cpp index 6c14c676b..1d52f616b 100644 --- a/gst.so/src/action.cpp +++ b/gst.so/src/action.cpp @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2025 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -60,6 +60,7 @@ using std::regex; #define RVS_CONF_TARGET_STRESS_KEY "target_stress" #define RVS_CONF_TOLERANCE_KEY "tolerance" #define RVS_CONF_HOT_CALLS "hot_calls" +#define RVS_CONF_WARM_CALLS "warm_calls" #define RVS_CONF_MATRIX_SIZE_KEYA "matrix_size_a" #define RVS_CONF_MATRIX_SIZE_KEYB "matrix_size_b" #define RVS_CONF_MATRIX_SIZE_KEYC "matrix_size_c" @@ -94,7 +95,7 @@ using std::regex; #define TARGET_KEY "target" #define DTYPE_KEY "dtype" -#define GST_DEFAULT_RAMP_INTERVAL 5000 +#define GST_DEFAULT_RAMP_INTERVAL 0 #define GST_DEFAULT_LOG_INTERVAL 1000 #define GST_DEFAULT_MAX_VIOLATIONS 0 #define GST_DEFAULT_TOLERANCE 0.05 @@ -102,6 +103,7 @@ using std::regex; #define GST_DEFAULT_MATRIX_SIZE 5760 #define GST_DEFAULT_MATRIX_INIT "default" #define GST_DEFAULT_HOT_CALLS 1 +#define GST_DEFAULT_WARM_CALLS 1 #define GST_DEFAULT_TRANS_A 0 #define GST_DEFAULT_TRANS_B 1 #define GST_DEFAULT_ALPHA_VAL 1 @@ -123,9 +125,7 @@ using std::regex; #define GST_DEFAULT_STRIDE_D 0 #define GST_DEFAULT_BLAS_SOURCE "rocblas" #define GST_DEFAULT_COMPUTE_TYPE "fp32_r" - -#define RVS_DEFAULT_PARALLEL false -#define RVS_DEFAULT_DURATION 0 +#define GST_DEFAULT_DURATION 0 #define GST_NO_COMPATIBLE_GPUS "No AMD compatible GPU found!" @@ -162,15 +162,16 @@ gst_action::~gst_action() { * @return true if no error occured, false otherwise */ bool gst_action::do_gpu_stress_test(map gst_gpus_device_index) { - size_t k = 0; + + uint64_t k = 0; + vector workers(gst_gpus_device_index.size()); + for (;;) { - unsigned int i = 0; if (property_wait != 0) // delay gst execution sleep(property_wait); - vector workers(gst_gpus_device_index.size()); - map::iterator it; + size_t i = 0; // all worker instances have the same json settings GSTWorker::set_use_json(bjson); @@ -191,6 +192,7 @@ bool gst_action::do_gpu_stress_test(map gst_gpus_device_index) { workers[i].set_target_stress(gst_target_stress); workers[i].set_tolerance(gst_tolerance); workers[i].set_gst_hot_calls(gst_hot_calls); + workers[i].set_gst_warm_calls(gst_warm_calls); workers[i].set_matrix_size_a(gst_matrix_size_a); workers[i].set_matrix_size_b(gst_matrix_size_b); workers[i].set_matrix_size_c(gst_matrix_size_c); @@ -255,7 +257,20 @@ bool gst_action::do_gpu_stress_test(map gst_gpus_device_index) { } } - return rvs::lp::Stopping() ? false : true; + if (rvs::lp::Stopping()) { + return false; + } + else { + + for (size_t i = 0; i < gst_gpus_device_index.size(); i++) { + if(false == workers[i].get_result()) { + return false; + } + } + } + + /* Action passed */ + return true; } /** @@ -348,6 +363,14 @@ bool gst_action::get_all_gst_config_keys(void) { bsts = false; } + error = property_get_int(RVS_CONF_WARM_CALLS, &gst_warm_calls, GST_DEFAULT_WARM_CALLS); + if (error == 1) { + msg = "invalid '" + + std::string(RVS_CONF_WARM_CALLS) + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + error = property_get_int(RVS_CONF_MATRIX_SIZE_KEYA, &gst_matrix_size_a, GST_DEFAULT_MATRIX_SIZE); if (error == 1) { msg = "invalid '" + @@ -568,6 +591,13 @@ bool gst_action::get_all_gst_config_keys(void) { bsts = false; } + if (property_get_int(RVS_CONF_DURATION_KEY, &property_duration, GST_DEFAULT_DURATION)) { + msg = "Invalid '" + std::string(RVS_CONF_DURATION_KEY) + + "' key"; + rvs::lp::Err(msg, module_name, action_name); + bsts = false; + } + /* If operation and data type both not set, default to sgemm */ if ((gst_ops_type == GST_DEFAULT_OPS_TYPE) && (gst_data_type == GST_DEFAULT_OPS_TYPE)) { gst_ops_type = "sgemm"; @@ -623,7 +653,7 @@ int gst_action::get_num_amd_gpu_devices(void) { rvs::lp::AddString(json_root_node, "ERROR", GST_NO_COMPATIBLE_GPUS); rvs::lp::LogRecordFlush(json_root_node, rvs::logerror); } - return 0; + return -1; } return hip_num_gpu_devices; } diff --git a/gst.so/src/gst_worker.cpp b/gst.so/src/gst_worker.cpp index 36ecd8f67..5e4b91636 100644 --- a/gst.so/src/gst_worker.cpp +++ b/gst.so/src/gst_worker.cpp @@ -124,7 +124,7 @@ void GSTWorker::hit_max_gflops(int *error, string *err_description) { gst_log_interval_time; double seconds_elapsed = 0, curr_gflops; uint16_t num_sgemm_ops_log_interval = 0; - uint64_t millis_sgemm_ops; + uint64_t micros_sgemm_ops; string msg; *error = 0; @@ -138,7 +138,7 @@ void GSTWorker::hit_max_gflops(int *error, string *err_description) { gst_end_time = std::chrono::system_clock::now(); if (time_diff(gst_end_time, gst_start_time) >= - NMAX_MS_GPU_RUN_PEAK_PERFORMANCE) + NMAX_MS_GPU_RUN_PEAK_PERFORMANCE * 1000u) break; if (copy_matrix) { @@ -151,7 +151,7 @@ void GSTWorker::hit_max_gflops(int *error, string *err_description) { } // run GEMM operation - if (!gpu_blas->run_blas_gemm(false)) + if (!gpu_blas->run_blas_gemm(1)) continue; // failed to run the GEMM operation // Waits for GEMM operation to complete @@ -161,10 +161,10 @@ void GSTWorker::hit_max_gflops(int *error, string *err_description) { num_sgemm_ops_log_interval++; gst_end_time = std::chrono::system_clock::now(); - millis_sgemm_ops = time_diff(gst_end_time, gst_log_interval_time); - if (millis_sgemm_ops >= log_interval) { + micros_sgemm_ops = time_diff(gst_end_time, gst_log_interval_time); + if (micros_sgemm_ops >= log_interval * 1000u) { // compute the GFLOPS - seconds_elapsed = static_cast (millis_sgemm_ops) / 1000; + seconds_elapsed = static_cast (micros_sgemm_ops) / 1000000; if (seconds_elapsed != 0) { curr_gflops = static_cast(gpu_blas->gemm_gflop_count() * num_sgemm_ops_log_interval) / seconds_elapsed; @@ -194,7 +194,7 @@ bool GSTWorker::do_gst_ramp(int *error, string *err_description) { gst_last_sgemm_end_time; double seconds_elapsed, curr_gflops, dyn_delay_target_stress; uint16_t num_sgemm_ops = 0, num_sgemm_ops_log_interval = 0; - uint64_t millis_sgemm_ops, millis_last_sgemm; + uint64_t micros_sgemm_ops, micros_last_sgemm; uint16_t proc_delay = 0; uint64_t start_time, end_time; double timetakenforoneiteration, gflops_interval; @@ -202,10 +202,10 @@ bool GSTWorker::do_gst_ramp(int *error, string *err_description) { // make sure that the ramp_interval & duration are not less than // NMAX_MS_GPU_RUN_PEAK_PERFORMANCE (e.g.: 1000) - if (run_duration_ms < NMAX_MS_GPU_RUN_PEAK_PERFORMANCE) + if (run_duration_ms > 0 && run_duration_ms < NMAX_MS_GPU_RUN_PEAK_PERFORMANCE) run_duration_ms += NMAX_MS_GPU_RUN_PEAK_PERFORMANCE; - if (ramp_interval < NMAX_MS_GPU_RUN_PEAK_PERFORMANCE) + if (ramp_interval > 0 && ramp_interval < NMAX_MS_GPU_RUN_PEAK_PERFORMANCE) ramp_interval += NMAX_MS_GPU_RUN_PEAK_PERFORMANCE; // stage 1. setup rvs blas @@ -221,6 +221,8 @@ bool GSTWorker::do_gst_ramp(int *error, string *err_description) { // the delay which gives the SGEMM frequency will be dynamically computed delay_target_stress = 0; + bool ramp_single_shot = (ramp_interval == 0); + gst_start_time = std::chrono::system_clock::now(); gst_log_interval_time = std::chrono::system_clock::now(); gst_start_gflops_time = std::chrono::system_clock::now(); @@ -230,10 +232,12 @@ bool GSTWorker::do_gst_ramp(int *error, string *err_description) { if (rvs::lp::Stopping()) return false; - gst_end_time = std::chrono::system_clock::now(); - if (time_diff(gst_end_time, gst_start_time) > - ramp_interval - NMAX_MS_GPU_RUN_PEAK_PERFORMANCE) - return false; + if (!ramp_single_shot) { + gst_end_time = std::chrono::system_clock::now(); + if (time_diff(gst_end_time, gst_start_time) > + (ramp_interval - NMAX_MS_GPU_RUN_PEAK_PERFORMANCE) * 1000u) + return false; + } gst_last_sgemm_start_time = std::chrono::system_clock::now(); @@ -252,7 +256,7 @@ bool GSTWorker::do_gst_ramp(int *error, string *err_description) { start_time = gpu_blas->get_time_us(); // run GEMM operation - if(!gpu_blas->run_blas_gemm(false)) { + if(!gpu_blas->run_blas_gemm(gst_warm_calls)) { *err_description = GST_BLAS_ERROR; *error = 1; @@ -273,30 +277,30 @@ bool GSTWorker::do_gst_ramp(int *error, string *err_description) { //Converting microseconds to seconds timetakenforoneiteration = (end_time - start_time)/1e6; - gflops_interval = gpu_blas->gemm_gflop_count()/timetakenforoneiteration; + gflops_interval = gpu_blas->gemm_gflop_count() * gst_warm_calls / timetakenforoneiteration; gst_last_sgemm_end_time = std::chrono::system_clock::now(); - millis_last_sgemm = + micros_last_sgemm = time_diff(gst_last_sgemm_end_time, gst_last_sgemm_start_time); if (static_cast( - (1000 * gpu_blas->gemm_gflop_count()) / + (1000000 * gpu_blas->gemm_gflop_count()) / target_stress) < - millis_last_sgemm) { + micros_last_sgemm) { // last SGEMM timed-out (it took more than it should) dyn_delay_target_stress = 1; } - num_sgemm_ops++; - num_sgemm_ops_log_interval++; + num_sgemm_ops += gst_warm_calls; + num_sgemm_ops_log_interval += gst_warm_calls; gst_end_time = std::chrono::system_clock::now(); - millis_sgemm_ops = + micros_sgemm_ops = time_diff(gst_end_time, gst_start_gflops_time); - if (millis_sgemm_ops >= NMAX_MS_SGEMM_OPS_RAMP_SUB_INTERVAL) { + if (micros_sgemm_ops >= NMAX_MS_SGEMM_OPS_RAMP_SUB_INTERVAL * 1000u) { // compute the GFLOPS seconds_elapsed = static_cast - (millis_sgemm_ops) / 1000; + (micros_sgemm_ops) / 1000000; if (seconds_elapsed > 0) { curr_gflops = static_cast( gpu_blas->gemm_gflop_count() * @@ -305,7 +309,7 @@ bool GSTWorker::do_gst_ramp(int *error, string *err_description) { target_stress + target_stress * tolerance/2) { ramp_actual_time = time_diff(gst_end_time, gst_start_time) + - NMAX_MS_GPU_RUN_PEAK_PERFORMANCE; + NMAX_MS_GPU_RUN_PEAK_PERFORMANCE * 1000u; delay_target_stress /= num_sgemm_ops; return true; } @@ -317,12 +321,12 @@ bool GSTWorker::do_gst_ramp(int *error, string *err_description) { gst_start_gflops_time = std::chrono::system_clock::now(); } - millis_sgemm_ops = + micros_sgemm_ops = time_diff(gst_end_time, gst_log_interval_time); - if (millis_sgemm_ops >= log_interval) { + if (micros_sgemm_ops >= log_interval * 1000u) { // compute the GFLOPS seconds_elapsed = static_cast - (millis_sgemm_ops) / 1000; + (micros_sgemm_ops) / 1000000; if (seconds_elapsed > 0) { curr_gflops = static_cast( @@ -334,6 +338,9 @@ bool GSTWorker::do_gst_ramp(int *error, string *err_description) { num_sgemm_ops_log_interval = 0; gst_log_interval_time = std::chrono::system_clock::now(); } + + if (ramp_single_shot) + return gflops_interval >= target_stress * (1.0 - tolerance); } return false; @@ -345,7 +352,6 @@ bool GSTWorker::do_gst_ramp(int *error, string *err_description) { */ void GSTWorker::check_target_stress(double gflops_interval) { string msg; - bool result; rvs::action_result_t action_result; char gpuid_buff[12]; auto desc = action_descriptor{action_name, MODULE_NAME,gpu_id}; @@ -370,7 +376,7 @@ void GSTWorker::check_target_stress(double gflops_interval) { if (bjson) log_to_json(desc ,rvs::logresults, TARGET_KEY, std::to_string(static_cast(target_stress)), - DTYPE_KEY, gst_ops_type, + DTYPE_KEY, !gst_data_type.empty() ? gst_data_type : gst_ops_type, "gflops", std::to_string(static_cast(gflops_interval)), "pass", result ? "true" : "false"); } @@ -434,11 +440,11 @@ bool GSTWorker::do_gst_stress_test(int *error, std::string *err_description) { uint32_t num_gemm_ops = 0; auto desc = action_descriptor{action_name, MODULE_NAME, gpu_id}; - uint64_t total_milliseconds, log_interval_milliseconds; + uint64_t total_microseconds, log_interval_microseconds; double start_time, end_time; double seconds_elapsed, gflops_interval; double timetakenforoneiteration; - double timetakenforniterations; + double timetakenforniterations = 0; string msg; std::chrono::time_point gst_start_time, gst_end_time, gst_log_interval_time; @@ -471,7 +477,7 @@ bool GSTWorker::do_gst_stress_test(int *error, std::string *err_description) { start_time = gpu_blas->get_time_us(); // launch GEMM operation - if(!gpu_blas->run_blas_gemm(true)) { + if(!gpu_blas->run_blas_gemm(gst_hot_calls)) { *err_description = GST_BLAS_ERROR; *error = 1; @@ -494,14 +500,14 @@ bool GSTWorker::do_gst_stress_test(int *error, std::string *err_description) { num_gemm_ops += gst_hot_calls; gst_end_time = std::chrono::system_clock::now(); - total_milliseconds = time_diff(gst_end_time, gst_start_time); + total_microseconds = time_diff(gst_end_time, gst_start_time); - log_interval_milliseconds = time_diff(gst_end_time, + log_interval_microseconds = time_diff(gst_end_time, gst_log_interval_time); - if (log_interval_milliseconds >= log_interval && num_gemm_ops > 0) { + if ((log_interval_microseconds >= log_interval * 1000u || 0 == run_duration_ms) && num_gemm_ops > 0) { - seconds_elapsed = static_cast (log_interval_milliseconds) / 1000; + seconds_elapsed = static_cast (log_interval_microseconds) / 1000000; if (seconds_elapsed != 0) { @@ -556,11 +562,11 @@ bool GSTWorker::do_gst_stress_test(int *error, std::string *err_description) { msg = "[" + action_name + "] " + MODULE_NAME + " " + std::to_string(gpu_id) + " " + GST_START_MSG + " " + - " Execution time in milliseconds :" + std::to_string(total_milliseconds) + + " Execution time in microseconds :" + std::to_string(total_microseconds) + " run_duration_ms :" + std::to_string(run_duration_ms); rvs::lp::Log(msg, rvs::logtrace); - if (total_milliseconds >= run_duration_ms) + if (0 == run_duration_ms || total_microseconds >= run_duration_ms * 1000u) break; } @@ -622,87 +628,45 @@ void GSTWorker::run() { std::to_string(gpu_id) + " " + " GST ramp completed for interval :" + " " + std::to_string(ramp_interval); rvs::lp::Log(msg, rvs::loginfo); - if (run_duration_ms > 0) { - gst_test_passed = do_gst_stress_test(&error, &err_description); - // check if stop signal was received - if (rvs::lp::Stopping()) - return; - if (error) { - // GPU didn't complete the test (HIP/rocBlas error(s) occurred) - string msg = "[" + action_name + "] " + MODULE_NAME + " " + - std::to_string(gpu_id) + " " + err_description; - rvs::lp::Log(msg, rvs::logerror); - if (bjson) - log_to_json(desc, rvs::logerror,"err", err_description); + gst_test_passed = do_gst_stress_test(&error, &err_description); + // check if stop signal was received + if (rvs::lp::Stopping()) + return; - action_result.state = rvs::actionstate::ACTION_COMPLETED; - action_result.status = rvs::actionstatus::ACTION_FAILED; - action_result.output = msg.c_str(); - action.action_callback(&action_result); + if (error) { + // GPU didn't complete the test (HIP/rocBlas error(s) occurred) + string msg = "[" + action_name + "] " + MODULE_NAME + " " + + std::to_string(gpu_id) + " " + err_description; + rvs::lp::Log(msg, rvs::logerror); + if (bjson) + log_to_json(desc, rvs::logerror,"err", err_description); - return; - } + action_result.state = rvs::actionstate::ACTION_COMPLETED; + action_result.status = rvs::actionstatus::ACTION_FAILED; + action_result.output = msg.c_str(); + action.action_callback(&action_result); + + return; } - log_interval_gflops(max_gflops); check_target_stress(max_gflops); } -void GSTWorker::log_gst_test_result(bool gst_test_passed) { -} /** - * @brief logs the GST test result - * @param gst_test_passed true if test succeeded, false otherwise - -void GSTWorker::log_gst_test_result(bool gst_test_passed) { - string msg; - - double flops_per_op = (2 * (static_cast(gpu_blas->get_m())/1000) * - (static_cast(gpu_blas->get_n())/1000) * - (static_cast(gpu_blas->get_k())/1000)); - msg = "[" + action_name + "] " + MODULE_NAME + " " + - std::to_string(gpu_id) + " " + GST_MAX_GFLOPS_OUTPUT_KEY + ": " + - std::to_string(max_gflops) + " " + GST_FLOPS_PER_OP_OUTPUT_KEY + ": " + - std::to_string(flops_per_op) + "x1e9" + " " + - GST_BYTES_COPIED_PER_OP_OUTPUT_KEY + ": " + - std::to_string(gpu_blas->get_bytes_copied_per_op()) + - " " + GST_TRY_OPS_PER_SEC_OUTPUT_KEY + ": "+ - std::to_string(target_stress / gpu_blas->gemm_gflop_count()) + - " " ; - rvs::lp::Log(msg, rvs::logresults); - - log_to_json(GST_MAX_GFLOPS_OUTPUT_KEY, std::to_string(max_gflops), - rvs::loginfo); - log_to_json(GST_FLOPS_PER_OP_OUTPUT_KEY, std::to_string(flops_per_op) + - "x1e9", rvs::loginfo); - log_to_json(GST_BYTES_COPIED_PER_OP_OUTPUT_KEY, - std::to_string(gpu_blas->get_bytes_copied_per_op()), - rvs::loginfo); - log_to_json(GST_TRY_OPS_PER_SEC_OUTPUT_KEY, - std::to_string(target_stress / gpu_blas->gemm_gflop_count()), - rvs::loginfo); - log_to_json(GST_PASS_KEY, (gst_test_passed ? - GST_RESULT_PASS_MESSAGE : GST_RESULT_FAIL_MESSAGE), - rvs::logresults); -} -*/ -/** - * @brief computes the difference (in milliseconds) between 2 points in time + * @brief computes the difference (in microseconds) between 2 points in time * @param t_end second point in time * @param t_start first point in time - * @return time difference in milliseconds + * @return time difference in microseconds */ uint64_t GSTWorker::time_diff( std::chrono::time_point t_end, std::chrono::time_point t_start) { - auto milliseconds = std::chrono::duration_cast( + auto microseconds = std::chrono::duration_cast( t_end - t_start); - return milliseconds.count(); + return microseconds.count(); } - - /** * @brief extends the usleep for more than 1000000us * @param microseconds us to sleep diff --git a/gst.so/src/rvs_module.cpp b/gst.so/src/rvs_module.cpp index 9f285d70d..b5ef86c07 100644 --- a/gst.so/src/rvs_module.cpp +++ b/gst.so/src/rvs_module.cpp @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2025 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -73,7 +73,6 @@ extern "C" int rvs_module_init(void* pMi) { } extern "C" int rvs_module_terminate(void) { - cleanup_logs(); amdsmi_shut_down(); return 0; } diff --git a/iet.so/CMakeLists.txt b/iet.so/CMakeLists.txt index 61b289e09..41a9098b0 100644 --- a/iet.so/CMakeLists.txt +++ b/iet.so/CMakeLists.txt @@ -163,7 +163,7 @@ include_directories(./ ../ ${AMD_SMI_INC_DIR} ${ROCBLAS_INC_DIR} ${ROCR_INC_DIR} # Add directories to look for library files to link link_directories(${RVS_LIB_DIR} ${ROCR_LIB_DIR} ${ROCBLAS_LIB_DIR} ${AMD_SMI_LIB_DIR} ${ASAN_LIB_PATH} ${HIPRAND_LIB_DIR} ${ROCRAND_LIB_DIR} ${HIPBLASLT_LIB_DIR}) ## additional libraries -set (PROJECT_LINK_LIBS rvslib libpthread.so libpci.so libm.so) +set (PROJECT_LINK_LIBS rvslib libpthread.so ${LIBPCI_TARGET} libm.so) set(SOURCES src/rvs_module.cpp src/action.cpp src/iet_worker.cpp ) diff --git a/iet.so/include/action.h b/iet.so/include/action.h index 04c3c0fbc..bea483fde 100644 --- a/iet.so/include/action.h +++ b/iet.so/include/action.h @@ -25,13 +25,6 @@ #ifndef IET_SO_INCLUDE_ACTION_H_ #define IET_SO_INCLUDE_ACTION_H_ -#ifdef __cplusplus -extern "C" { -#endif -#include -#ifdef __cplusplus -} -#endif #include #include @@ -178,7 +171,8 @@ class iet_action: public rvs::actionbase { */ int get_all_selected_gpus(void); - bool do_edp_test(std::map iet_gpus_device_index); + bool do_edp_test(map iet_gpus_device_index, + std::vector& mcm_type); }; #endif // IET_SO_INCLUDE_ACTION_H_ diff --git a/iet.so/include/iet_worker.h b/iet.so/include/iet_worker.h index 1105d734e..5189dc366 100644 --- a/iet.so/include/iet_worker.h +++ b/iet.so/include/iet_worker.h @@ -293,8 +293,13 @@ class IETWorker : public rvs::ThreadBase { //! sets gemm output data type void set_iet_out_data_type(std::string out_data_type) { iet_out_data_type = out_data_type; } + //! set GPU MCM (Multi-Chip Module) type - Primary/Secondary + void set_mcm_type(mcm_type_t _mcm_type) { mcm_type = _mcm_type; } + //! BLAS callback static void blas_callback (bool status, void *user_data); + //! get worker job result + bool get_result(void) { return result; } protected: virtual void run(void); bool do_gpu_init_training(int gpuIdx, uint64_t matrix_size, std::string iet_ops_type); @@ -421,6 +426,11 @@ class IETWorker : public rvs::ThreadBase { bool nt_loads; //! gemm output data type std::string iet_out_data_type; + //! GPU MCM (Multi-Chip Module) type - Primary/Secondary + mcm_type_t mcm_type; + + //! Worker job result + bool result; }; #endif // IET_SO_INCLUDE_IET_WORKER_H_ diff --git a/iet.so/src/action.cpp b/iet.so/src/action.cpp index 0ea42d623..553e62d61 100644 --- a/iet.so/src/action.cpp +++ b/iet.so/src/action.cpp @@ -34,13 +34,6 @@ #include #include -#ifdef __cplusplus -extern "C" { -#endif -#include -#ifdef __cplusplus -} -#endif #include #define __HIP_PLATFORM_HCC__ @@ -162,7 +155,7 @@ iet_action::iet_action() { * @brief class destructor */ iet_action::~iet_action() { - property.clear(); + property.clear(); } /** @@ -487,155 +480,162 @@ bool iet_action::get_all_iet_config_keys(void) { */ void iet_action::hip_to_smi_indices(void) { - int hip_num_gpu_devices; - hipGetDeviceCount(&hip_num_gpu_devices); - // map this to smi as only these are visible - uint32_t smi_num_devices; - uint64_t val_ui64; - - std::map smi_map; - - smi_map = rvs::get_smi_pci_map(); - - for (int i = 0; i < hip_num_gpu_devices; i++) { - // get GPU device properties - //hipDeviceProp_t props; - //hipGetDeviceProperties(&props, i); - unsigned int pDom, pBus, pDev, pFun; - getBDF(i, pDom, pBus, pDev, pFun); - // compute device location_id (needed to match this device - // with one of those found while querying the pci bus - uint64_t hip_dev_location_id = ( ( ((uint64_t)pDom & 0xffff ) << 32) | - (((uint64_t) pBus & 0xff ) << 8) | (((uint64_t)pDev & 0x1f ) << 3)| ((uint64_t)pFun ) ); - - if(smi_map.find(hip_dev_location_id) != smi_map.end()){ - hip_to_smi_idxs.insert({i, smi_map[hip_dev_location_id]}); - } + + int hip_num_gpu_devices; + hipGetDeviceCount(&hip_num_gpu_devices); + // map this to smi as only these are visible + uint32_t smi_num_devices; + uint64_t val_ui64; + + std::map smi_map; + + smi_map = rvs::get_smi_pci_map(); + + for (int i = 0; i < hip_num_gpu_devices; i++) { + // get GPU device properties + //hipDeviceProp_t props; + //hipGetDeviceProperties(&props, i); + unsigned int pDom, pBus, pDev, pFun; + getBDF(i, pDom, pBus, pDev, pFun); + // compute device location_id (needed to match this device + // with one of those found while querying the pci bus + uint64_t hip_dev_location_id = ( ( ((uint64_t)pDom & 0xffff ) << 32) | + (((uint64_t) pBus & 0xff ) << 8) | (((uint64_t)pDev & 0x1f ) << 3)| ((uint64_t)pFun ) ); + + if(smi_map.find(hip_dev_location_id) != smi_map.end()){ + hip_to_smi_idxs.insert({i, smi_map[hip_dev_location_id]}); } + } } - /** * @brief runs the edp test * @return true if no error occured, false otherwise */ -bool iet_action::do_edp_test(map iet_gpus_device_index) { - std::string msg; - uint32_t dev_idx = 0; - size_t k = 0; - int gpuId; - bool gpu_masking = false; // if HIP_VISIBLE_DEVICES is set, this will be true - int hip_num_gpu_devices; - hipGetDeviceCount(&hip_num_gpu_devices); - - vector workers(iet_gpus_device_index.size()); - for (;;) { - unsigned int i = 0; - map::iterator it; - - if (property_wait != 0) // delay iet execution - sleep(property_wait); - - // map hip indexes to smi indexes - hip_to_smi_indices(); - - IETWorker::set_use_json(bjson); - for (it = iet_gpus_device_index.begin(); it != iet_gpus_device_index.end(); ++it) { - if(hip_to_smi_idxs.find(it->first) != hip_to_smi_idxs.end()){ - workers[i].set_smi_device_handle(hip_to_smi_idxs[it->first]); - } else{ - workers[i].set_smi_device_handle(nullptr);// this must not happen - msg = "[" + action_name + "] " + MODULE_NAME + " " + std::to_string(i) + " has no handle"; - rvs::lp::Log(msg, rvs::logerror); - } - gpuId = it->second; - // set worker thread params - workers[i].set_name(action_name); - workers[i].set_action(*this); - workers[i].set_gpu_id(it->second); - workers[i].set_gpu_device_index(it->first); - workers[i].set_pwr_device_id(dev_idx++); - workers[i].set_run_wait_ms(property_wait); - workers[i].set_run_duration_ms(property_duration); - workers[i].set_ramp_interval(iet_ramp_interval); - workers[i].set_log_interval(property_log_interval); - workers[i].set_sample_interval(iet_sample_interval); - workers[i].set_max_violations(iet_max_violations); - workers[i].set_target_power(iet_target_power); - workers[i].set_tolerance(iet_tolerance); - workers[i].set_matrix_size(iet_matrix_size); - workers[i].set_matrix_size_a(iet_matrix_size_a); - workers[i].set_matrix_size_b(iet_matrix_size_b); - workers[i].set_matrix_size_c(iet_matrix_size_c); - workers[i].set_iet_ops_type(iet_ops_type); - workers[i].set_iet_data_type(iet_data_type); - workers[i].set_matrix_transpose_a(iet_trans_a); - workers[i].set_matrix_transpose_b(iet_trans_b); - workers[i].set_alpha_val(iet_alpha_val); - workers[i].set_beta_val(iet_beta_val); - workers[i].set_lda_offset(iet_lda_offset); - workers[i].set_ldb_offset(iet_ldb_offset); - workers[i].set_ldc_offset(iet_ldc_offset); - workers[i].set_ldd_offset(iet_ldd_offset); - workers[i].set_tp_flag(iet_tp_flag); - workers[i].set_bw_workload(iet_bw_workload); - workers[i].set_cp_workload(iet_cp_workload); - workers[i].set_hot_calls(iet_hot_calls); - workers[i].set_matrix_init(iet_matrix_init); - workers[i].set_gemm_mode(iet_gemm_mode); - workers[i].set_batch_size(iet_batch_size); - workers[i].set_stride_a(iet_stride_a); - workers[i].set_stride_b(iet_stride_b); - workers[i].set_stride_c(iet_stride_c); - workers[i].set_stride_d(iet_stride_d); - workers[i].set_blas_source(iet_blas_source); - workers[i].set_compute_type(iet_compute_type); - workers[i].set_wg_count(iet_wg_count); - workers[i].set_nt_loads(iet_nt_loads); - workers[i].set_iet_out_data_type(iet_out_data_type); - i++; - } +bool iet_action::do_edp_test(map iet_gpus_device_index, + std::vector& mcm_type) { + + std::string msg; + uint32_t dev_idx = 0; + size_t k = 0; + unsigned int i = 0; + int gpuId; + bool gpu_masking = false; // if HIP_VISIBLE_DEVICES is set, this will be true + int hip_num_gpu_devices; + hipGetDeviceCount(&hip_num_gpu_devices); + vector workers(iet_gpus_device_index.size()); + + for (;;) { + map::iterator it; + + if (property_wait != 0) // delay iet execution + sleep(property_wait); + + // map hip indexes to smi indexes + hip_to_smi_indices(); + + IETWorker::set_use_json(bjson); + for (it = iet_gpus_device_index.begin(); it != iet_gpus_device_index.end(); ++it) { + if(hip_to_smi_idxs.find(it->first) != hip_to_smi_idxs.end()){ + workers[i].set_smi_device_handle(hip_to_smi_idxs[it->first]); + } else{ + workers[i].set_smi_device_handle(nullptr);// this must not happen + msg = "[" + action_name + "] " + MODULE_NAME + " " + std::to_string(i) + " has no handle"; + rvs::lp::Log(msg, rvs::logerror); + } + gpuId = it->second; + // set worker thread params + workers[i].set_name(action_name); + workers[i].set_action(*this); + workers[i].set_gpu_id(it->second); + workers[i].set_gpu_device_index(it->first); + workers[i].set_pwr_device_id(dev_idx++); + workers[i].set_run_wait_ms(property_wait); + workers[i].set_run_duration_ms(property_duration); + workers[i].set_ramp_interval(iet_ramp_interval); + workers[i].set_log_interval(property_log_interval); + workers[i].set_sample_interval(iet_sample_interval); + workers[i].set_max_violations(iet_max_violations); + workers[i].set_target_power(iet_target_power); + workers[i].set_tolerance(iet_tolerance); + workers[i].set_matrix_size(iet_matrix_size); + workers[i].set_matrix_size_a(iet_matrix_size_a); + workers[i].set_matrix_size_b(iet_matrix_size_b); + workers[i].set_matrix_size_c(iet_matrix_size_c); + workers[i].set_iet_ops_type(iet_ops_type); + workers[i].set_iet_data_type(iet_data_type); + workers[i].set_matrix_transpose_a(iet_trans_a); + workers[i].set_matrix_transpose_b(iet_trans_b); + workers[i].set_alpha_val(iet_alpha_val); + workers[i].set_beta_val(iet_beta_val); + workers[i].set_lda_offset(iet_lda_offset); + workers[i].set_ldb_offset(iet_ldb_offset); + workers[i].set_ldc_offset(iet_ldc_offset); + workers[i].set_ldd_offset(iet_ldd_offset); + workers[i].set_tp_flag(iet_tp_flag); + workers[i].set_bw_workload(iet_bw_workload); + workers[i].set_cp_workload(iet_cp_workload); + workers[i].set_hot_calls(iet_hot_calls); + workers[i].set_matrix_init(iet_matrix_init); + workers[i].set_gemm_mode(iet_gemm_mode); + workers[i].set_batch_size(iet_batch_size); + workers[i].set_stride_a(iet_stride_a); + workers[i].set_stride_b(iet_stride_b); + workers[i].set_stride_c(iet_stride_c); + workers[i].set_stride_d(iet_stride_d); + workers[i].set_blas_source(iet_blas_source); + workers[i].set_compute_type(iet_compute_type); + workers[i].set_wg_count(iet_wg_count); + workers[i].set_nt_loads(iet_nt_loads); + workers[i].set_iet_out_data_type(iet_out_data_type); + workers[i].set_mcm_type(mcm_type[i]); + + i++; + } - if (property_parallel) { - for (i = 0; i < iet_gpus_device_index.size(); i++) - workers[i].start(); - // join threads - for (i = 0; i < iet_gpus_device_index.size(); i++) - workers[i].join(); - - } else { - for (i = 0; i < iet_gpus_device_index.size(); i++) { - workers[i].start(); - workers[i].join(); - - // check if stop signal was received - if (rvs::lp::Stopping()) { - return false; - } - } - } + if (property_parallel) { + for (i = 0; i < iet_gpus_device_index.size(); i++) + workers[i].start(); + // join threads + for (i = 0; i < iet_gpus_device_index.size(); i++) + workers[i].join(); + } else { + for (i = 0; i < iet_gpus_device_index.size(); i++) { + workers[i].start(); + workers[i].join(); - msg = "[" + action_name + "] " + MODULE_NAME + " " + std::to_string(gpuId) + " Shutting down smi "; - rvs::lp::Log(msg, rvs::loginfo); + // check if stop signal was received + if (rvs::lp::Stopping()) { + return false; + } + } + } + msg = "[" + action_name + "] " + MODULE_NAME + " " + std::to_string(gpuId) + " Shutting down smi "; + rvs::lp::Log(msg, rvs::loginfo); - // check if stop signal was received - if (rvs::lp::Stopping()) - return false; + // check if stop signal was received + if (rvs::lp::Stopping()) + return false; - if (property_count == ++k) { - break; - } + if (property_count == ++k) { + break; } + } + msg = "[" + action_name + "] " + MODULE_NAME + " " + std::to_string(gpuId) + " Done with iet test "; + rvs::lp::Log(msg, rvs::loginfo); - msg = "[" + action_name + "] " + MODULE_NAME + " " + std::to_string(gpuId) + " Done with iet test "; - rvs::lp::Log(msg, rvs::loginfo); + sleep(1000); - sleep(1000); + for (i = 0; i < iet_gpus_device_index.size(); i++) { + if(false == workers[i].get_result()) + return false; + } - return true; + /* IET action passed */ + return true; } /** @@ -643,64 +643,65 @@ bool iet_action::do_edp_test(map iet_gpus_device_index) { * @return run number of GPUs */ int iet_action::get_num_amd_gpu_devices(void) { - int hip_num_gpu_devices; - string msg; - - hipGetDeviceCount(&hip_num_gpu_devices); - return hip_num_gpu_devices; -} + int hip_num_gpu_devices; + string msg; + hipGetDeviceCount(&hip_num_gpu_devices); + return hip_num_gpu_devices; +} /** * @brief gets all selected GPUs and starts the worker threads * @return run result */ int iet_action::get_all_selected_gpus(void) { - int hip_num_gpu_devices; - bool amd_gpus_found = false; - map iet_gpus_device_index; - std::string msg; - std::stringstream msg_stream; - - hipGetDeviceCount(&hip_num_gpu_devices); - if (hip_num_gpu_devices < 1) - return hip_num_gpu_devices; - - // find compatible GPUs to run edp tests - amd_gpus_found = fetch_gpu_list(hip_num_gpu_devices, iet_gpus_device_index, - property_device, property_device_id, property_device_all, - property_device_index, property_device_index_all, true); // MCM checks - if(!amd_gpus_found){ - msg = "No devices match criteria from the test configuation."; - rvs::lp::Log(msg, rvs::logerror); - if (bjson) { - unsigned int sec; - unsigned int usec; - rvs::lp::get_ticks(&sec, &usec); - void *json_root_node = rvs::lp::LogRecordCreate(MODULE_NAME, - action_name.c_str(), rvs::logerror, sec, usec, true); - if (!json_root_node) { - // log the error - string msg = std::string(JSON_CREATE_NODE_ERROR); - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - return -1; - } - - rvs::lp::AddString(json_root_node, "ERROR", "No AMD compatible GPU found!"); - rvs::lp::LogRecordFlush(json_root_node, rvs::logerror); - } - - return 0; // no GPUs is not error + + int hip_num_gpu_devices; + bool amd_gpus_found = false; + map iet_gpus_device_index; + std::string msg; + std::stringstream msg_stream; + std::vector mcm_type; + + hipGetDeviceCount(&hip_num_gpu_devices); + if (hip_num_gpu_devices < 1) + return -1; + + // find compatible GPUs to run edp tests + amd_gpus_found = fetch_gpu_list(hip_num_gpu_devices, iet_gpus_device_index, + property_device, property_device_id, property_device_all, + property_device_index, property_device_index_all, true, &mcm_type); // MCM checks + if(!amd_gpus_found){ + msg = "No devices match criteria from the test configuation."; + rvs::lp::Log(msg, rvs::logerror); + if (bjson) { + unsigned int sec; + unsigned int usec; + rvs::lp::get_ticks(&sec, &usec); + void *json_root_node = rvs::lp::LogRecordCreate(MODULE_NAME, + action_name.c_str(), rvs::logerror, sec, usec, true); + if (!json_root_node) { + // log the error + string msg = std::string(JSON_CREATE_NODE_ERROR); + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + return -1; + } + + rvs::lp::AddString(json_root_node, "ERROR", "No AMD compatible GPU found!"); + rvs::lp::LogRecordFlush(json_root_node, rvs::logerror); } - int iet_res = 0; - if(do_edp_test(iet_gpus_device_index)) - iet_res = 0; - else - iet_res = -1; - // append end node to json - return iet_res; + return -1; + } + + int iet_res = 0; + if(do_edp_test(iet_gpus_device_index, mcm_type)) + iet_res = 0; + else + iet_res = -1; + // append end node to json + return iet_res; } diff --git a/iet.so/src/iet_worker.cpp b/iet.so/src/iet_worker.cpp index 16b65ea16..b7e9a6ad4 100644 --- a/iet.so/src/iet_worker.cpp +++ b/iet.so/src/iet_worker.cpp @@ -133,7 +133,7 @@ void IETWorker::computeThread(void) { while ((duration < run_duration_ms) && (endtest == false)) { // run GEMM operation - if(!gpu_blas->run_blas_gemm(true)) { + if(!gpu_blas->run_blas_gemm(iet_hot_calls)) { endtest = true; break; } @@ -170,7 +170,6 @@ bool IETWorker::do_iet_power_stress(void) { float cur_power_value = 0; float totalpower = 0; float max_power = 0; - bool result = true; bool start = true; rvs::action_result_t action_result; char gpuid_buff[12]; @@ -255,12 +254,23 @@ bool IETWorker::do_iet_power_stress(void) { result = true; } else { - msg = "[" + action_name + "] " + MODULE_NAME + " " + - std::to_string(gpu_id) + " " + " Average power could not meet the target power \ - in the given interval, increase the duration and try again, \ - Average power is :" + " " + std::to_string(max_power); - rvs::lp::Log(msg, rvs::loginfo); - result = false; + + if(mcm_type == mcm_type_t::PRIMARY) { + msg = "[" + action_name + "] " + MODULE_NAME + " " + + std::to_string(gpu_id) + " " + " Average power could not meet the target power \ + in the given interval, increase the duration and try again, \ + Average power is :" + " " + std::to_string(max_power); + rvs::lp::Log(msg, rvs::loginfo); + result = false; + } + else { + /* For secondary MCM, there is no power reporting - so by default considering it as pass */ + msg = "[" + action_name + "] " + MODULE_NAME + " " + + std::to_string(gpu_id) + " " + " No power reporting present for \ + secondary MCM, considering the test as pass by default" ; + rvs::lp::Log(msg, rvs::loginfo); + result = true; + } } if (IETWorker::bjson) diff --git a/iet.so/src/rvs_module.cpp b/iet.so/src/rvs_module.cpp index 5511248dd..137575d2e 100644 --- a/iet.so/src/rvs_module.cpp +++ b/iet.so/src/rvs_module.cpp @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2025 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -74,7 +74,6 @@ extern "C" int rvs_module_init(void* pMi) { } extern "C" int rvs_module_terminate(void) { - cleanup_logs(); amdsmi_shut_down(); return 0; } diff --git a/include/gpu_util.h b/include/gpu_util.h index f66b29635..0ede856f0 100644 --- a/include/gpu_util.h +++ b/include/gpu_util.h @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -44,8 +44,10 @@ extern void gpu_get_all_domain_id(std::vector* pgpus_domain_id, std::map , uint16_t>& pgpus_dom_loc_map); extern bool gpu_check_if_mcm_die (int idx); extern int gpu_hip_to_smi_hdl(int hip_index, amdsmi_processor_handle* smi_index); +extern int gpu_hip_to_node(int hip_index, int* node); extern void gpu_get_all_pci_bdf(std::vector& ppci_bdf); extern bool gpu_check_if_gpu_indexes (const std::vector &idx); +extern std::string gpu_get_platform_name (void); namespace rvs { @@ -75,6 +77,7 @@ class gpulist { static int domlocation2gpu(const uint16_t domainID, const uint16_t LocationID, uint16_t* pGPUID); static int node2bdf(const uint16_t NodeID, std::string& pPciBDF); + static std::string gpu_get_platform_name (void); protected: //! Array of GPU location IDs static std::vector location_id; diff --git a/include/rvs_blas.h b/include/rvs_blas.h index b2e83a2ce..ea10fb390 100644 --- a/include/rvs_blas.h +++ b/include/rvs_blas.h @@ -107,7 +107,7 @@ class rvs_blas { void generate_random_matrix_data(void); bool copy_data_to_gpu(void); template bool copy_data_to_gpu(void); - bool run_blas_gemm(bool hot_call); + bool run_blas_gemm(uint64_t num_calls); bool is_gemm_op_complete(void); bool validate_gemm(bool self_check, bool accu_check, double &self_error, double &accu_error); void set_gemm_error(uint64_t _error_freq, uint64_t _error_count); @@ -339,8 +339,8 @@ class rvs_blas { (datatype == "fp6_e2m3_r") ? HIP_R_6F_E2M3 : (datatype == "i8_r") ? HIP_R_8I : (datatype == "fp8_r") ? HIP_R_8F_E4M3_FNUZ : // FP8-FNUZ - (datatype == "fp8_e4m3_r") ? HIP_R_8F_E4M3 : // FP8-OCP E4M3 - (datatype == "fp8_e5m2_r") ? HIP_R_8F_E5M2 : // FP8-OCP E5M2 + (datatype == "fp8_e4m3_r" || datatype == "mxfp8_e4m3_r") ? HIP_R_8F_E4M3 : // FP8-OCP E4M3 + (datatype == "fp8_e5m2_r" || datatype == "mxfp8_e5m2_r") ? HIP_R_8F_E5M2 : // FP8-OCP E5M2 (datatype == "bf16_r") ? HIP_R_16BF : (datatype == "fp16_r") ? HIP_R_16F : (datatype == "fp32_r") ? HIP_R_32F : diff --git a/include/rvs_key_def.h b/include/rvs_key_def.h index 8e272e2a2..ec28ebf80 100644 --- a/include/rvs_key_def.h +++ b/include/rvs_key_def.h @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -45,6 +45,16 @@ #define RVS_CONF_HOT_CALLS_KEY "hot_calls" #define RVS_CONF_WARM_CALLS_KEY "warm_calls" #define RVS_CONF_B2B_KEY "b2b" +#define RVS_CONF_EXECUTOR_KEY "executor" +#define RVS_CONF_SUBEXECUTOR_KEY "subexecutor" +#define RVS_CONF_TRANSFER_METHOD_KEY "transfer_method" +#define RVS_CONF_TRANSFERBENCH_TEST_KEY "transferbench_test" +#define RVS_CONF_A2A_MODE_KEY "a2a_mode" +#define RVS_CONF_A2A_DIRECT_KEY "a2a_direct" +#define RVS_CONF_A2A_LOCAL_KEY "a2a_local" +#define RVS_CONF_A2A_NUM_GPUS_KEY "a2a_num_gpus" +#define RVS_CONF_USE_REMOTE_READ_KEY "use_remote_read" +#define RVS_CONF_GFX_UNROLL_KEY "gfx_unroll" #define DEFAULT_LOG_INTERVAL (1000u) #define DEFAULT_DURATION (10000u) @@ -53,6 +63,18 @@ #define DEFAULT_HOT_CALLS (1u) #define DEFAULT_WARM_CALLS (1u) #define DEFAULT_B2B false +#define DEFAULT_TRANSFER_METHOD "native" +#define DEFAULT_TRANSFERBENCH_TEST "p2p" +#define DEFAULT_EXECUTOR "gfx" +#define DEFAULT_SUBEXECUTOR (1u) +#define DEFAULT_A2A_MODE (0u) +#define DEFAULT_A2A_DIRECT (1u) +#define DEFAULT_A2A_LOCAL (0u) +#define DEFAULT_A2A_NUM_GPUS (0u) +#define DEFAULT_USE_REMOTE_READ (0u) +#define DEFAULT_GFX_UNROLL (4u) +#define DEFAULT_SRC_MEMORY "null" +#define DEFAULT_DST_MEMORY "null" #define YAML_DEVICE_PROPERTY_ERROR "Error while parsing property" #define YAML_DEVICEID_PROPERTY_ERROR "Error while parsing "\ diff --git a/include/rvs_util.h b/include/rvs_util.h index 2da4358e0..432145fe6 100644 --- a/include/rvs_util.h +++ b/include/rvs_util.h @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2025 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -34,6 +34,21 @@ using std::map; +//! GPU MCM (Multi-Chip Module) types - Primary/Secondary +enum class mcm_type_t { + PRIMARY = 0, + SECONDARY = 1 +}; + +struct device_info { + std::string bus; + std::string name; + int32_t node_id; + int32_t gpu_id; + int32_t device_id; + uint64_t bdfId;// this is pcie location id to uniquely identify device +}; + struct action_descriptor{ std::string action_name; std::string module_name; @@ -44,6 +59,46 @@ extern bool is_positive_integer(const std::string& str_val); extern std::vector str_split(const std::string& str_val, const std::string& delimiter); +/** + * ROCm install **root** for runtime: if built with FETCH_ROCMPATH_FROM_ROCMCORE, + * tries getROCmInstallPath (rocm-core) first, then the ROCM_PATH environment + * variable, then the build-time ROCM_PATH string. Otherwise: getenv(ROCM_PATH) + * first, then the build-time macro. Use this to resolve relocatable or + * non-default ROCm locations instead of relying on the prebuilt #define alone. + */ +std::string rvs_get_rocm_install_path_string(void); + +/** + * @brief RVS data root (share/.../rocm-validation-suite). + * + * Path derived from the running rvs binary on Linux, else build-time + * RVS_DATA_ROOT. Use -c to load a config from an arbitrary location. + */ +std::string rvs_get_rvs_data_root_string(void); +/** + * @brief RVS module library directory (.../lib/rvs). + * + * Same prefix resolution as rvs_get_rvs_data_root_string(). Final fallback + * when relative module search paths in rvsmodule.cpp do not find the .so. + */ +std::string rvs_get_rvs_modules_lib_dir_string(void); + +/** + * @brief Validate a module .so before dlopen(). + * + * On Linux: regular file, not world-writable; group-writable only when owned by + * the caller (local builds). When euid is 0, reject group/world writable and + * require root ownership. + * + * @param path Candidate .so path. + * @param err_msg Optional failure reason. + * @param canonical_path Optional canonical path on success. + * @return true if checks pass, false otherwise. + */ +bool rvs_verify_module_so_for_dlopen(const std::string& path, + std::string* err_msg, + std::string* canonical_path = nullptr); + /** * Convert array of strings into array of signed integers of type T * @param sArr input string @@ -130,13 +185,18 @@ int rvs_util_parse(const std::string& buff, void *json_node_create(std::string module_name, std::string action_name, int log_level); + bool fetch_gpu_list(int hip_num_gpu_devices, map& gpus_device_index, const std::vector& property_device, const int& property_device_id, bool property_device_all, const std::vector& property_device_index, - bool property_device_index_all, bool mcm_check = false); + bool property_device_index_all, bool mcm_check = false, + std::vector* mcm_type = nullptr); + void getBDF(int idx ,unsigned int& domain,unsigned int& bus,unsigned int& device,unsigned int& function); -int display_gpu_info(void); +int display_gpu_info(std::vector); void *json_list_create(std::string lname, int log_level); +std::vector get_gpu_info (void); +std::string get_gpu_name (void); template void log_to_json(action_descriptor desc, int log_level, KVPairs... key_values ) { diff --git a/include/rvshsa.h b/include/rvshsa.h index 5882117fc..b13c6c81f 100644 --- a/include/rvshsa.h +++ b/include/rvshsa.h @@ -157,7 +157,7 @@ class hsa { int SendTraffic(uint32_t SrcNode, uint32_t DstNode, size_t Size, bool bidirectional, bool b2b, uint32_t warm_calls, uint32_t hot_calls, - double* Duration); + double* Duration, uint32_t* NumTimed = nullptr); int GetPeerStatus(uint32_t SrcNode, uint32_t DstNode); int GetPeerStatusAgent(const AgentInformation& SrcAgent, diff --git a/include/rvsliblog.h b/include/rvsliblog.h index 174a0ac3e..8d4a2706d 100644 --- a/include/rvsliblog.h +++ b/include/rvsliblog.h @@ -50,6 +50,7 @@ typedef void (*t_cbAddNode)(void* Parent, void* Child); typedef void (*t_cbStop)(uint16_t flags); typedef bool (*t_cbStopping)(void); typedef void* (*t_cbJsonNamedListCreate)( const char* name, const int LogLevel); +typedef void* (*t_cbJsonNestedListCreate)( const char* name, const int LogLevel); typedef int (*t_rvs_module_err)(const char*, const char*, const char*); @@ -88,6 +89,7 @@ typedef struct tag_module_init { //! pointer to rvs::logger::Err() function t_rvs_module_err cbErr; t_cbJsonNamedListCreate cbJsonNamedListCreate; + t_cbJsonNestedListCreate cbJsonNestedListCreate; } T_MODULE_INIT; #ifdef __cplusplus diff --git a/include/rvsliblogger.h b/include/rvsliblogger.h index 35f83e85a..e6baae7d4 100644 --- a/include/rvsliblogger.h +++ b/include/rvsliblogger.h @@ -83,6 +83,7 @@ class logger { static int Err(const char *Message, const char *Module = nullptr, const char *Action = nullptr); static void* JsonNamedListCreate(const char* name, const int LogLevel); + static void* JsonNestedListCreate(const char* name, const int LogLevel); protected: static int ToFile(const std::string& Row , bool json = false); diff --git a/include/rvsloglp.h b/include/rvsloglp.h index 9290075c2..e76367d8f 100644 --- a/include/rvsloglp.h +++ b/include/rvsloglp.h @@ -88,6 +88,7 @@ class lp { static int Err(const std::string &Msg, const std::string &Module, const std::string &Action); static void* JsonNamedListCreate(const char* name, const int LogLevel); + static void* JsonNestedListCreate(const char* name, const int LogLevel); protected: //! Module init structure passed through Initialize() method diff --git a/include/rvslognodelist.h b/include/rvslognodelist.h index 649311f9a..52130ae77 100644 --- a/include/rvslognodelist.h +++ b/include/rvslognodelist.h @@ -44,7 +44,7 @@ namespace rvs { */ class LogListNode : virtual public LogNode { public: - explicit LogListNode(const char* Name, int LogLevel, const LogNodeBase* Parent = nullptr); + explicit LogListNode(const char* Name, int LogLevel, bool nested = false, const LogNodeBase* Parent = nullptr); virtual ~LogListNode(); virtual std::string ToJson(const std::string& Lead = ""); @@ -58,6 +58,7 @@ class LogListNode : virtual public LogNode { protected: int Level; + bool IsNested; }; } // namespace rvs diff --git a/mem.so/CMakeLists.txt b/mem.so/CMakeLists.txt index 978f550e7..04edd2f89 100644 --- a/mem.so/CMakeLists.txt +++ b/mem.so/CMakeLists.txt @@ -144,7 +144,7 @@ include_directories(./ ../ ${ROCR_INC_DIR} ${HIP_INC_DIR} ${HIPRAND_INC_DIR} ${R # Add directories to look for library files to link link_directories(${RVS_LIB_DIR} ${ROCR_LIB_DIR} ${HIP_LIB_DIR} ${ROCBLAS_LIB_DIR} ${ASAN_LIB_PATH} ${AMD_SMI_LIB_DIR} ${HIPRAND_LIB_DIR} ${ROCRAND_LIB_DIR}) ## additional libraries -set (PROJECT_LINK_LIBS rvslib libpthread.so libpci.so libm.so) +set (PROJECT_LINK_LIBS rvslib libpthread.so libm.so) ## define source files set(SOURCES src/rvs_module.cpp src/action.cpp src/rvs_memtest.cpp src/rvs_memworker.cpp) diff --git a/mem.so/include/action.h b/mem.so/include/action.h index 0c03afa09..7fe1957f6 100644 --- a/mem.so/include/action.h +++ b/mem.so/include/action.h @@ -25,13 +25,6 @@ #ifndef MEM_SO_INCLUDE_ACTION_H_ #define MEM_SO_INCLUDE_ACTION_H_ -#ifdef __cplusplus -extern "C" { -#endif -#include -#ifdef __cplusplus -} -#endif #include #include diff --git a/mem.so/src/action.cpp b/mem.so/src/action.cpp index 98e0a5a2e..ef3528ba0 100644 --- a/mem.so/src/action.cpp +++ b/mem.so/src/action.cpp @@ -82,7 +82,8 @@ mem_action::~mem_action() { * @return true if no error occured, false otherwise */ bool mem_action::do_mem_stress_test(map mem_gpus_device_index) { - size_t k = 0; + + uint64_t k = 0; string msg; for (;;) { @@ -264,7 +265,7 @@ int mem_action::get_num_amd_gpu_devices(void) { rvs::lp::AddString(json_root_node, "ERROR", MEM_NO_COMPATIBLE_GPUS); rvs::lp::LogRecordFlush(json_root_node); } - return 0; + return -1; } return hip_num_gpu_devices; } diff --git a/mem.so/src/rvs_memtest.cpp b/mem.so/src/rvs_memtest.cpp index 9e4deffd8..d8793d9ec 100644 --- a/mem.so/src/rvs_memtest.cpp +++ b/mem.so/src/rvs_memtest.cpp @@ -147,10 +147,9 @@ unsigned int error_checking(const std::string& pmsg, unsigned int blockidx) hipMemset(ptCntOfError, 0, sizeof(unsigned int)); hipMemset((void*)&ptFailedAdress[0], 0, sizeof(unsigned long)*MAX_ERR_RECORD_COUNT);; hipMemset((void*)&ptExpectedValue[0], 0, sizeof(unsigned long)*MAX_ERR_RECORD_COUNT);; - hipMemset((void*)&ptCurrentValue[0], 0, sizeof(unsigned long)*MAX_ERR_RECORD_COUNT);; + hipMemset((void*)&ptCurrentValue[0], 0, sizeof(unsigned long)*MAX_ERR_RECORD_COUNT);; - hipDeviceReset(); - exit(ERR_BAD_STATE); + return numOfErrors; } @@ -1639,14 +1638,18 @@ void allocate_small_mem(void) void free_small_mem(void) { - //Initialize memory hipFree((void*)ptCntOfError); + ptCntOfError = nullptr; hipFree((void*)ptFailedAdress); + ptFailedAdress = nullptr; hipFree((void*)ptExpectedValue); + ptExpectedValue = nullptr; hipFree((void*)ptCurrentValue); + ptCurrentValue = nullptr; hipFree((void*)ptValueOfSecondRead); + ptValueOfSecondRead = nullptr; } diff --git a/mem.so/src/rvs_memworker.cpp b/mem.so/src/rvs_memworker.cpp index a4ebf6c71..efdacfbe0 100644 --- a/mem.so/src/rvs_memworker.cpp +++ b/mem.so/src/rvs_memworker.cpp @@ -233,6 +233,7 @@ void MemWorker::run() { std::to_string(tot_num_blocks); rvs::lp::Log(msg, rvs::logtrace); + free_small_mem(); return; } @@ -280,6 +281,14 @@ void MemWorker::run() { run_tests(ptr, tot_num_blocks); + if (useMappedMemory) { + hipHostFree(mappedHostPtr); + mappedHostPtr = nullptr; + } else { + hipFree(ptr); + ptr = nullptr; + } + free_small_mem(); } diff --git a/mem.so/src/rvs_module.cpp b/mem.so/src/rvs_module.cpp index 2c0859ccb..bdbf6d643 100644 --- a/mem.so/src/rvs_module.cpp +++ b/mem.so/src/rvs_module.cpp @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2025 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -84,7 +84,6 @@ extern "C" int rvs_module_init(void* pMi) { } extern "C" int rvs_module_terminate(void) { - cleanup_logs(); amdsmi_shut_down(); return 0; } diff --git a/pbqt.so/CMakeLists.txt b/pbqt.so/CMakeLists.txt index ccaea42cb..8fd80ceff 100644 --- a/pbqt.so/CMakeLists.txt +++ b/pbqt.so/CMakeLists.txt @@ -1,6 +1,6 @@ ################################################################################ ## -## Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. +## Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. ## ## MIT LICENSE: ## Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -133,11 +133,14 @@ if(NOT EXISTS ${ROCR_LIB_DIR}/${CORE_RUNTIME_LIBRARY}.so) endif() ## define include directories -include_directories(./ ../ pci ${ROCR_INC_DIR} ${HIPRAND_INC_DIR} ${ROCRAND_INC_DIR}) +include_directories(./ ../ pci ${ROCR_INC_DIR} ${HIPRAND_INC_DIR} ${ROCRAND_INC_DIR} ${TRANSFERBENCH_INC_DIR}) +if(TRANSFERBENCH_IBVERBS_INC_DIR) + include_directories(${TRANSFERBENCH_IBVERBS_INC_DIR}) +endif() # Add directories to look for library files to link link_directories(${RVS_LIB_DIR} ${ROCR_LIB_DIR} ${ASAN_LIB_PATH} ${AMD_SMI_LIB_DIR} ${HIPRAND_LIB_DIR} ${ROCRAND_LIB_DIR}) ## additional libraries -set (PROJECT_LINK_LIBS rvslib libpthread.so libpci.so libm.so) +set (PROJECT_LINK_LIBS rvslib libpthread.so ${LIBPCI_TARGET} libm.so) ## define source files set(SOURCES src/rvs_module.cpp src/action.cpp src/action_run.cpp @@ -149,6 +152,9 @@ set_target_properties(${RVS_TARGET} PROPERTIES SUFFIX .so.${LIB_VERSION_STRING} LIBRARY_OUTPUT_DIRECTORY ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}) target_link_libraries(${RVS_TARGET} ${PROJECT_LINK_LIBS} ${CORE_RUNTIME_TARGET}) +if(TRANSFERBENCH_IBVERBS_INC_DIR) + target_link_libraries(${RVS_TARGET} ${CMAKE_DL_LIBS}) +endif() add_dependencies(${RVS_TARGET} rvslib) add_custom_command(TARGET ${RVS_TARGET} POST_BUILD diff --git a/pbqt.so/include/action.h b/pbqt.so/include/action.h index a30af3270..5f55d1fc5 100644 --- a/pbqt.so/include/action.h +++ b/pbqt.so/include/action.h @@ -1,133 +1,162 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#ifndef PBQT_SO_INCLUDE_ACTION_H_ -#define PBQT_SO_INCLUDE_ACTION_H_ - -#include -#include -#include - -#include -#include -#include -#include -#include -#include - -#include - -#include "hsa/hsa.h" -#include "hsa/hsa_ext_amd.h" - -#include "include/rvsactionbase.h" - -using namespace std::chrono; - - -class pbqtworker; - -enum class pbqt_json_data_t { - PBQT_THROUGHPUT = 0, - PBQT_LINK_TYPE = 1 -}; - -/** - * @class pbqt_action - * @ingroup PBQT - * - * @brief PBQT action implementation class - * - * Derives from rvs::actionbase and implements actual action functionality - * in its run() method. - * - */ -class pbqt_action : public rvs::actionbase { - public: - pbqt_action(); - virtual ~pbqt_action(); - - virtual int run(void); - - protected: - bool get_all_pbqt_config_keys(void); - - // PBQT specific config keys - bool property_get_peers(int *error); - void property_get_test_bandwidth(int *error); -// void property_get_log_interval(int *error); - void property_get_bidirectional(int *error); - - //! 'true' if "all" is found under "peer" key for this action - bool prop_peer_device_all_selected; - //! array of peer GPU IDs to be used in data trasfers - std::vector prop_peers; - //! deviceid of peer GPUs - uint32_t prop_peer_deviceid; - //! 'true' if bandwidth test is to be executed for verified peers - bool prop_test_bandwidth; - //! 'true' if bidirectional data transfer is required - bool prop_bidirectional; - //! list of test block sizes - std::vector block_size; - //! set to 'true' if the default block sizes are to be used - bool b_block_size_all; - //! test block size for back-to-back transfers - uint32_t b2b_block_size; - //! link type - int link_type; - - std::string link_type_string; - - protected: - int is_peer(uint16_t Src, uint16_t Dst); - int create_threads(); - int destroy_threads(); - - int run_single(); - int run_parallel(); - - int print_running_average(); - int print_running_average(pbqtworker* pWorker); - - int print_final_average(); - - //! 'true' for the duration of test - bool brun; - - - void* json_base_node(int log_level); - void json_add_kv(void *json_node, const std::string &key, const std::string &value); - void json_to_file(void *json_node,int log_level); - void log_json_data(std::string srcnode, std::string dstnode, - int log_level, std::string conn, std::string throughput); - - private: - void do_running_average(void); - void do_final_average(void); - - std::vector test_array; -}; - -#endif // PBQT_SO_INCLUDE_ACTION_H_ +/******************************************************************************** + * + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. + * + * MIT LICENSE: + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies + * of the Software, and to permit persons to whom the Software is furnished to do + * so, subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + * + *******************************************************************************/ +#ifndef PBQT_SO_INCLUDE_ACTION_H_ +#define PBQT_SO_INCLUDE_ACTION_H_ + +#include +#include +#include + +#include +#include +#include +#include +#include +#include + +#include + +#include "hsa/hsa.h" +#include "hsa/hsa_ext_amd.h" + +#include "include/rvsactionbase.h" + +using namespace std::chrono; + + +class pbqtworker; + +enum class pbqt_json_data_t { + PBQT_THROUGHPUT = 0, + PBQT_LINK_TYPE = 1 +}; + +/** + * @class pbqt_action + * @ingroup PBQT + * + * @brief PBQT action implementation class + * + * Derives from rvs::actionbase and implements actual action functionality + * in its run() method. + * + */ +class pbqt_action : public rvs::actionbase { + public: + pbqt_action(); + virtual ~pbqt_action(); + + virtual int run(void); + + protected: + bool get_all_pbqt_config_keys(void); + + // PBQT specific config keys + bool property_get_peers(int *error); + void property_get_test_bandwidth(int *error); +// void property_get_log_interval(int *error); + void property_get_bidirectional(int *error); + + //! 'true' if "all" is found under "peer" key for this action + bool prop_peer_device_all_selected; + //! array of peer GPU IDs to be used in data trasfers + std::vector prop_peers; + //! deviceid of peer GPUs + uint32_t prop_peer_deviceid; + //! 'true' if bandwidth test is to be executed for verified peers + bool prop_test_bandwidth; + //! 'true' if bidirectional data transfer is required + bool prop_bidirectional; + //! list of test block sizes + std::vector block_size; + //! set to 'true' if the default block sizes are to be used + bool b_block_size_all; + //! test block size for back-to-back transfers + uint32_t b2b_block_size; + //! link type + int link_type; + + std::string link_type_string; + //! Number of warm calls (transfer iterations) before bandwidth calculation (hot calls) + //! to ignore few intial transfers for the bandwidth to settle + uint32_t warm_calls; + //! Number of hot calls (transfer iterations) for bandwidth calculation after warm calls + uint32_t hot_calls; + //! 'true' if back-to-back transfers enabled + bool b2b; + + //! transfer method - TransferBench or Native + std::string transfer_method; + //! transferbench test type - p2p or alltoall + std::string transferbench_test; + //! transfer executor to use - GPU or SDMA + std::string executor; + //! No. of subexecutors + uint32_t subexecutor; + + //! alltoall mode: 0=copy, 1=read-only, 2=write-only + uint32_t a2a_mode; + //! alltoall direct-only: 1=only direct XGMI links, 0=full all-to-all + uint32_t a2a_direct; + //! alltoall local: 1=include self-transfers, 0=exclude + uint32_t a2a_local; + //! number of GPUs for alltoall (0=all detected) + uint32_t a2a_num_gpus; + //! remote read: 1=use DST as executor, 0=use SRC + uint32_t use_remote_read; + //! GFX kernel unroll factor + uint32_t gfx_unroll; + + protected: + int is_peer(uint16_t Src, uint16_t Dst); + int create_threads(); + int destroy_threads(); + + int run_single(); + int run_parallel(); + + int print_running_average(); + int print_running_average(pbqtworker* pWorker); + + int print_final_average(); + + //! 'true' for the duration of test + bool brun; + + + void* json_base_node(int log_level); + void json_add_kv(void *json_node, const std::string &key, const std::string &value); + void json_to_file(void *json_node,int log_level); + void log_json_data(std::string srcnode, std::string dstnode, + int log_level, std::string conn, std::string throughput); + + private: + void do_running_average(void); + void do_final_average(void); + + std::vector test_array; +}; + +#endif // PBQT_SO_INCLUDE_ACTION_H_ diff --git a/pbqt.so/include/worker.h b/pbqt.so/include/worker.h index 1338c54c0..b5fd53f25 100644 --- a/pbqt.so/include/worker.h +++ b/pbqt.so/include/worker.h @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -43,6 +43,20 @@ * */ +struct peer_pair_t { + uint16_t srcnode; + uint16_t dstnode; + std::string conn_type; +}; + +struct gpu_pair_bw_t { + int srcGpu; + int dstGpu; + int srcNode; + int dstNode; + double avgBandwidthGbPerSec; +}; + namespace rvs { class hsa; } @@ -87,6 +101,33 @@ class pbqtworker : public rvs::ThreadBase { const std::string& get_conn_type(){ return conn_type; } + void set_hot_calls(uint32_t _hot_calls) { hot_calls = _hot_calls; } + //! Set warm calls + void set_warm_calls(uint32_t _warm_calls) { warm_calls = _warm_calls; } + //! Set b2b + void set_b2b(bool _b2b) { b2b = _b2b; } + //! Set transfer method + void set_transfer_method(std::string _transfer_method) { transfer_method = _transfer_method; } + //! Set executor + void set_executor(std::string _executor) { executor = _executor; } + //! Set subexecutor + void set_subexecutor(uint32_t _subexecutor) { subexecutor = _subexecutor; } + //! Set transferbench test type (p2p or alltoall) + void set_transferbench_test(std::string _test) { transferbench_test = _test; } + //! Set alltoall mode (0=copy, 1=read-only, 2=write-only) + void set_a2a_mode(uint32_t val) { a2a_mode = val; } + //! Set alltoall direct-only flag (1=only direct XGMI links) + void set_a2a_direct(uint32_t val) { a2a_direct = val; } + //! Set alltoall local flag (1=include self-transfers) + void set_a2a_local(uint32_t val) { a2a_local = val; } + //! Set number of GPUs for alltoall (0=all detected) + void set_a2a_num_gpus(uint32_t val) { a2a_num_gpus = val; } + //! Set remote read flag (1=use DST as executor instead of SRC) + void set_use_remote_read(uint32_t val) { use_remote_read = val; } + //! Set GFX kernel unroll factor + void set_gfx_unroll(uint32_t val) { gfx_unroll = val; } + //! Get per GPU-pair bandwidth results from alltoall + const std::vector& get_gpu_pair_bw() const { return gpu_pair_bw; } protected: virtual void run(void); @@ -127,9 +168,42 @@ class pbqtworker : public rvs::ThreadBase { //! total number of transfers uint16_t transfer_num; + //! hot calls + uint32_t hot_calls; + //! warm calls + uint32_t warm_calls; + //! 'true' if back-to-back transfers enabled + bool b2b; + + //! transfer method - TransferBench or Native + std::string transfer_method; + //! transferbench test type - p2p or alltoall + std::string transferbench_test; + //! transfer executor to use - GPU or SDMA + std::string executor; + //! No. of subexecutors + uint32_t subexecutor; + //! list of test block sizes std::vector block_size; std::string conn_type; + + //! alltoall mode: 0=copy, 1=read-only, 2=write-only + uint32_t a2a_mode; + //! alltoall direct-only: 1=only direct XGMI links, 0=full all-to-all + uint32_t a2a_direct; + //! alltoall local: 1=include self-transfers, 0=exclude + uint32_t a2a_local; + //! number of GPUs for alltoall (0=all detected) + uint32_t a2a_num_gpus; + //! remote read: 1=use DST as executor, 0=use SRC + uint32_t use_remote_read; + //! GFX kernel unroll factor + uint32_t gfx_unroll; + + //! per GPU-pair bandwidth results from alltoall + std::vector gpu_pair_bw; + //! synchronization mutex std::mutex cntmutex; }; diff --git a/pbqt.so/src/action.cpp b/pbqt.so/src/action.cpp index 465fcd8f7..c57456e36 100644 --- a/pbqt.so/src/action.cpp +++ b/pbqt.so/src/action.cpp @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2025 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -25,7 +25,6 @@ #include "include/action.h" extern "C" { -#include #include } #include @@ -36,6 +35,8 @@ extern "C" { #include #include #include +#include +#include #include "include/rvs_key_def.h" #include "include/pci_caps.h" @@ -223,6 +224,104 @@ bool pbqt_action::get_all_pbqt_config_keys(void) { link_type_string = "XGMI"; } + if (property_get(RVS_CONF_B2B_KEY, &b2b, DEFAULT_B2B)) { + msg = "invalid '" + std::string(RVS_CONF_B2B_KEY) + "' key"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + res = false; + } + + error = property_get_int(RVS_CONF_HOT_CALLS_KEY, &hot_calls, DEFAULT_HOT_CALLS); + if (error == 1) { + msg = "invalid '" + std::string(RVS_CONF_HOT_CALLS_KEY) + "' key"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + res = false; + } + + error = property_get_int(RVS_CONF_WARM_CALLS_KEY, &warm_calls, DEFAULT_WARM_CALLS); + if (error == 1) { + msg = "invalid '" + std::string(RVS_CONF_WARM_CALLS_KEY) + "' key"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + res = false; + } + + error = property_get(RVS_CONF_TRANSFER_METHOD_KEY, &transfer_method, DEFAULT_TRANSFER_METHOD); + if (error == 1) { + msg = "invalid '" + std::string(RVS_CONF_TRANSFER_METHOD_KEY) + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + res = false; + } + + error = property_get(RVS_CONF_TRANSFERBENCH_TEST_KEY, &transferbench_test, DEFAULT_TRANSFERBENCH_TEST); + if (error == 1) { + msg = "invalid '" + std::string(RVS_CONF_TRANSFERBENCH_TEST_KEY) + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + res = false; + } + + error = property_get(RVS_CONF_EXECUTOR_KEY, &executor, DEFAULT_EXECUTOR); + if (error == 1) { + msg = "invalid '" + std::string(RVS_CONF_EXECUTOR_KEY) + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + res = false; + } + + error = property_get_int(RVS_CONF_SUBEXECUTOR_KEY, &subexecutor, DEFAULT_SUBEXECUTOR); + if (error == 1) { + msg = "invalid '" + std::string(RVS_CONF_SUBEXECUTOR_KEY) + "' key"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + res = false; + } + + error = property_get_int(RVS_CONF_A2A_MODE_KEY, &a2a_mode, DEFAULT_A2A_MODE); + if (error == 1) { + msg = "invalid '" + std::string(RVS_CONF_A2A_MODE_KEY) + "' key"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + res = false; + } + + error = property_get_int(RVS_CONF_A2A_DIRECT_KEY, &a2a_direct, DEFAULT_A2A_DIRECT); + if (error == 1) { + msg = "invalid '" + std::string(RVS_CONF_A2A_DIRECT_KEY) + "' key"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + res = false; + } + + error = property_get_int(RVS_CONF_A2A_LOCAL_KEY, &a2a_local, DEFAULT_A2A_LOCAL); + if (error == 1) { + msg = "invalid '" + std::string(RVS_CONF_A2A_LOCAL_KEY) + "' key"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + res = false; + } + + error = property_get_int(RVS_CONF_A2A_NUM_GPUS_KEY, &a2a_num_gpus, DEFAULT_A2A_NUM_GPUS); + if (error == 1) { + msg = "invalid '" + std::string(RVS_CONF_A2A_NUM_GPUS_KEY) + "' key"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + res = false; + } + + error = property_get_int(RVS_CONF_USE_REMOTE_READ_KEY, &use_remote_read, DEFAULT_USE_REMOTE_READ); + if (error == 1) { + msg = "invalid '" + std::string(RVS_CONF_USE_REMOTE_READ_KEY) + "' key"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + res = false; + } + + error = property_get_int(RVS_CONF_GFX_UNROLL_KEY, &gfx_unroll, DEFAULT_GFX_UNROLL); + if (error == 1) { + msg = "invalid '" + std::string(RVS_CONF_GFX_UNROLL_KEY) + "' key"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + res = false; + } + + if(!hot_calls) { + hot_calls = DEFAULT_HOT_CALLS; + } + + if(!warm_calls) { + warm_calls = DEFAULT_WARM_CALLS; + } + return res; } @@ -288,14 +387,18 @@ int pbqt_action::create_threads() { std::vector gpu_id; std::vector gpu_idx; std::vector gpu_device_id; - uint16_t transfer_ix = 0; bool bmatch_found = false; char srcgpuid_buff[12]; char dstgpuid_buff[12]; - std::string pbconn; gpu_get_all_gpu_id(&gpu_id); gpu_get_all_gpu_idx(&gpu_idx); gpu_get_all_device_id(&gpu_device_id); + + std::vector valid_pairs; + + // --------------------------------------------------------------- + // Loop 1: Discover all valid srcnode-dstnode peer pairs + // --------------------------------------------------------------- for (size_t i = 0; i < gpu_id.size(); i++) { // all possible sources // filter out by source device id @@ -353,7 +456,7 @@ int pbqt_action::create_threads() { } RVSTRACE_ - // signal that at lease one matching src-dst combination + // signal that at least one matching src-dst combination // has been found: bmatch_found = true; @@ -361,7 +464,7 @@ int pbqt_action::create_threads() { // get NUMA nodes uint16_t srcnode; if (rvs::gpulist::gpu2node(gpu_id[i], &srcnode)) { - msg + "no node found for GPU ID " + std::to_string(gpu_id[i]); + msg = "no node found for GPU ID " + std::to_string(gpu_id[i]); rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); return -1; } @@ -427,40 +530,17 @@ int pbqt_action::create_threads() { if(distance == rvs::hsa::NO_CONN) { continue; // no point if no connection } + + std::string pbconn; if (0 != arr_linkinfo.size()) { - /* Log link type */ pbconn = arr_linkinfo[0].strtype; - //log_json_data(std::to_string(srcnode), std::to_string(gpu_id[j]), rvs::logresults, - // pbqt_json_data_t::PBQT_LINK_TYPE, arr_linkinfo[0].strtype); - /* Note: Assuming link type for all hops between GPUs are the same */ } RVSTRACE_ - // GPUs are peers, create transaction for them if (prop_test_bandwidth) { - RVSTRACE_ - pbqtworker* p = nullptr; - - transfer_ix += 1; - - p = new pbqtworker; - if (p == nullptr) { - RVSTRACE_ - msg = "internal error"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - return -1; - } - p->initialize(srcnode, dstnode, prop_bidirectional); - RVSTRACE_ - p->set_name(action_name); - p->set_stop_name(action_name); - p->set_transfer_ix(transfer_ix); - p->set_block_sizes(block_size); - p->set_conn_type(pbconn); - test_array.push_back(p); - }else{ - // no need to run bandwidth, just log interface info - log_json_data(std::to_string(gpu_id[srcnode]), std::to_string(gpu_id[j]), rvs::logresults, + valid_pairs.push_back({srcnode, dstnode, pbconn}); + } else { + log_json_data(std::to_string(gpu_id[i]), std::to_string(gpu_id[j]), rvs::logresults, pbconn, "NA" ); } @@ -492,8 +572,11 @@ int pbqt_action::create_threads() { } } + // --------------------------------------------------------------- + // Loop 2: Create pbqtworker for each valid srcnode-dstnode pair + // --------------------------------------------------------------- RVSTRACE_ - if (prop_test_bandwidth && test_array.size() < 1) { + if (prop_test_bandwidth && valid_pairs.empty()) { RVSTRACE_ std::string diag; if (bmatch_found) { @@ -525,6 +608,50 @@ int pbqt_action::create_threads() { return 0; } + uint16_t transfer_ix = 0; + bool alltoall_single = (transfer_method == "transferbench" && + transferbench_test == "alltoall"); + + for (const auto& pair : valid_pairs) { + RVSTRACE_ + pbqtworker* p = nullptr; + + transfer_ix += 1; + + p = new pbqtworker; + if (p == nullptr) { + RVSTRACE_ + msg = "internal error"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + return -1; + } + p->initialize(pair.srcnode, pair.dstnode, prop_bidirectional); + RVSTRACE_ + p->set_name(action_name); + p->set_stop_name(action_name); + p->set_transfer_ix(transfer_ix); + p->set_block_sizes(block_size); + p->set_conn_type(pair.conn_type); + p->set_b2b(b2b); + p->set_hot_calls(hot_calls); + p->set_warm_calls(warm_calls); + p->set_transfer_method(transfer_method); + p->set_transferbench_test(transferbench_test); + p->set_executor(executor); + p->set_subexecutor(subexecutor); + p->set_a2a_mode(a2a_mode); + p->set_a2a_direct(a2a_direct); + p->set_a2a_local(a2a_local); + p->set_a2a_num_gpus(a2a_num_gpus); + p->set_use_remote_read(use_remote_read); + p->set_gfx_unroll(gfx_unroll); + + test_array.push_back(p); + + if (alltoall_single) + break; + } + RVSTRACE_ for (auto it = test_array.begin(); it != test_array.end(); ++it) { RVSTRACE_ @@ -739,79 +866,202 @@ int pbqt_action::print_final_average() { rvs::action_result_t result; for (auto it = test_array.begin(); it != test_array.end(); ++it) { - (*it)->get_final_data(&src_node, &dst_node, &bidir, - ¤t_size, &duration); - if (duration) { - bandwidth = current_size/duration/1000 / 1000 / 1000; - if (bidir) { - bandwidth *=2; + if (transferbench_test != "alltoall") { + (*it)->get_final_data(&src_node, &dst_node, &bidir, + ¤t_size, &duration); + + if (duration) { + bandwidth = current_size/duration/1000 / 1000 / 1000; + if (bidir) { + bandwidth *=2; + } + snprintf( buff, sizeof(buff), "%.3f GBps", bandwidth); + } else { + snprintf( buff, sizeof(buff), "(not measured)"); } - snprintf( buff, sizeof(buff), "%.3f GBps", bandwidth); - } else { - snprintf( buff, sizeof(buff), "(not measured)"); - } - RVSTRACE_ - if (rvs::gpulist::node2gpu(src_node, &src_id)) { + + RVSTRACE_ + if (rvs::gpulist::node2gpu(src_node, &src_id)) { + RVSTRACE_ + std::string msg = "could not find GPU id for node " + + std::to_string(src_node); + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + return -1; + } + RVSTRACE_ + if (rvs::gpulist::node2gpu(dst_node, &dst_id)) { + RVSTRACE_ + std::string msg = "could not find GPU id for node " + + std::to_string(dst_node); + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + return -1; + } + + std::string src_pci_bdf; + if (rvs::gpulist::node2bdf(src_node, src_pci_bdf)) { RVSTRACE_ - std::string msg = "could not find GPU id for node " + + std::string msg = "could not find PCI BDF for node " + std::to_string(src_node); rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); return -1; } - RVSTRACE_ - if (rvs::gpulist::node2gpu(dst_node, &dst_id)) { + + std::string dst_pci_bdf; + if (rvs::gpulist::node2bdf(dst_node, dst_pci_bdf)) { RVSTRACE_ - std::string msg = "could not find GPU id for node " + + std::string msg = "could not find PCI BDF for node " + std::to_string(dst_node); rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); return -1; } - std::string src_pci_bdf; - if (rvs::gpulist::node2bdf(src_node, src_pci_bdf)) { - RVSTRACE_ - std::string msg = "could not find PCI BDF for node " + - std::to_string(src_node); - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - return -1; - } + transfer_ix = (*it)->get_transfer_ix(); + transfer_num = (*it)->get_transfer_num(); - std::string dst_pci_bdf; - if (rvs::gpulist::node2bdf(dst_node, dst_pci_bdf)) { - RVSTRACE_ - std::string msg = "could not find PCI BDF for node " + - std::to_string(dst_node); - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - return -1; + snprintf(transfer_buff, sizeof(transfer_buff), "%2d", transfer_ix); + snprintf(srcgpuid_buff, sizeof(srcgpuid_buff), "%5d", src_id); + snprintf(dstgpuid_buff, sizeof(dstgpuid_buff), "%5d", dst_id); + + msg = "[" + action_name + "] p2p-bandwidth[" + + transfer_buff + "/" + std::to_string(transfer_num) + "]" + + " [GPU:: " + std::to_string(src_node) + " - " + srcgpuid_buff + " - " + src_pci_bdf + "]" + + " [GPU:: " + std::to_string(dst_node) + " - " + dstgpuid_buff + " - " + dst_pci_bdf + "]" + + " bidirectional: " + std::string(bidir ? "true" : "false") + + " " + buff + " duration: " + std::to_string(duration) + " secs"; + + rvs::lp::Log(msg, rvs::logresults); + + result.state = rvs::actionstate::ACTION_RUNNING; + result.status = rvs::actionstatus::ACTION_SUCCESS; + result.output = msg.c_str(); + action_callback(&result); + + log_json_data(std::to_string(src_id), std::to_string(dst_id), rvs::logresults, + (*it)->get_conn_type(), buff); + + sleep(1); } + else { - transfer_ix = (*it)->get_transfer_ix(); - transfer_num = (*it)->get_transfer_num(); + /* AlltoAll final results */ - snprintf(transfer_buff, sizeof(transfer_buff), "%2d", transfer_ix); - snprintf(srcgpuid_buff, sizeof(srcgpuid_buff), "%5d", src_id); - snprintf(dstgpuid_buff, sizeof(dstgpuid_buff), "%5d", dst_id); + const auto& pair_bw = (*it)->get_gpu_pair_bw(); + std::map gpuAggregateBw; + std::set uniqueGpus; + double a2a_total_bandwidth = 0.0; - msg = "[" + action_name + "] p2p-bandwidth[" - + transfer_buff + "/" + std::to_string(transfer_num) + "]" - + " [GPU:: " + std::to_string(src_node) + " - " + srcgpuid_buff + " - " + src_pci_bdf + "]" - + " [GPU:: " + std::to_string(dst_node) + " - " + dstgpuid_buff + " - " + dst_pci_bdf + "]" - + " bidirectional: " + std::string(bidir ? "true" : "false") - + " " + buff + " duration: " + std::to_string(duration) + " secs"; + for (size_t p = 0; p < pair_bw.size(); p++) { + const auto& pb = pair_bw[p]; + + gpuAggregateBw[pb.srcGpu] += pb.avgBandwidthGbPerSec; + uniqueGpus.insert(pb.srcGpu); + uniqueGpus.insert(pb.dstGpu); + a2a_total_bandwidth += pb.avgBandwidthGbPerSec; + + uint16_t pb_src_id = 0, pb_dst_id = 0; + if (rvs::gpulist::node2gpu(pb.srcNode, &pb_src_id)) { + std::string errmsg = "could not find GPU id for node " + + std::to_string(pb.srcNode); + rvs::lp::Err(errmsg, MODULE_NAME_CAPS, action_name); + continue; + } + if (rvs::gpulist::node2gpu(pb.dstNode, &pb_dst_id)) { + std::string errmsg = "could not find GPU id for node " + + std::to_string(pb.dstNode); + rvs::lp::Err(errmsg, MODULE_NAME_CAPS, action_name); + continue; + } + + std::string pb_src_bdf; + if (rvs::gpulist::node2bdf(pb.srcNode, pb_src_bdf)) { + std::string errmsg = "could not find PCI BDF for node " + + std::to_string(pb.srcNode); + rvs::lp::Err(errmsg, MODULE_NAME_CAPS, action_name); + continue; + } + + std::string pb_dst_bdf; + if (rvs::gpulist::node2bdf(pb.dstNode, pb_dst_bdf)) { + std::string errmsg = "could not find PCI BDF for node " + + std::to_string(pb.dstNode); + rvs::lp::Err(errmsg, MODULE_NAME_CAPS, action_name); + continue; + } + + char pb_transfer_buff[8]; + char pb_srcgpuid_buff[12]; + char pb_dstgpuid_buff[12]; + char pb_bw_buff[128]; + + snprintf(pb_transfer_buff, sizeof(pb_transfer_buff), "%2zu", p + 1); + snprintf(pb_srcgpuid_buff, sizeof(pb_srcgpuid_buff), "%5d", pb_src_id); + snprintf(pb_dstgpuid_buff, sizeof(pb_dstgpuid_buff), "%5d", pb_dst_id); + snprintf(pb_bw_buff, sizeof(pb_bw_buff), "%.3f GBps", pb.avgBandwidthGbPerSec); + + msg = "[" + action_name + "] a2a-p2p-bandwidth[" + + pb_transfer_buff + "/" + std::to_string(pair_bw.size()) + "]" + + " [GPU:: " + std::to_string(pb.srcNode) + " - " + pb_srcgpuid_buff + " - " + pb_src_bdf + "]" + + " [GPU:: " + std::to_string(pb.dstNode) + " - " + pb_dstgpuid_buff + " - " + pb_dst_bdf + "]" + + " " + pb_bw_buff; + + rvs::lp::Log(msg, rvs::logresults); + + result.state = rvs::actionstate::ACTION_RUNNING; + result.status = rvs::actionstatus::ACTION_SUCCESS; + result.output = msg.c_str(); + action_callback(&result); + + log_json_data(std::to_string(pb_src_id), std::to_string(pb_dst_id), + rvs::logresults, (*it)->get_conn_type(), pb_bw_buff); + } + + if (!gpuAggregateBw.empty()) { - rvs::lp::Log(msg, rvs::logresults); + for (const auto& entry : gpuAggregateBw) { + int gpuNode = -1; + gpu_hip_to_node(entry.first, &gpuNode); - result.state = rvs::actionstate::ACTION_RUNNING; - result.status = rvs::actionstatus::ACTION_SUCCESS; - result.output = msg.c_str(); - action_callback(&result); + uint16_t gpu_id_val = 0; + std::string gpu_bdf; + rvs::gpulist::node2gpu(gpuNode, &gpu_id_val); + rvs::gpulist::node2bdf(gpuNode, gpu_bdf); - log_json_data(std::to_string(src_id), std::to_string(dst_id), rvs::logresults, - (*it)->get_conn_type(), buff); + char aggr_gpuid_buff[12]; + char aggr_bw_buff[128]; + snprintf(aggr_gpuid_buff, sizeof(aggr_gpuid_buff), "%5d", gpu_id_val); + snprintf(aggr_bw_buff, sizeof(aggr_bw_buff), "%.3f GBps", entry.second); - sleep(1); + msg = "[" + action_name + "] a2a-gpu-bandwidth" + + " [GPU:: " + std::to_string(gpuNode) + " - " + aggr_gpuid_buff + " - " + gpu_bdf + "]" + + " Aggregate peer bandwidth: " + aggr_bw_buff; + + rvs::lp::Log(msg, rvs::logresults); + + result.state = rvs::actionstate::ACTION_RUNNING; + result.status = rvs::actionstatus::ACTION_SUCCESS; + result.output = msg.c_str(); + action_callback(&result); + } + + char total_bw_buff[128]; + snprintf(total_bw_buff, sizeof(total_bw_buff), "%.3f GBps", + a2a_total_bandwidth); + + msg = "[" + action_name + "] a2a-bandwidth " + + "[" + std::to_string(uniqueGpus.size()) + " GPUs]" + + "[" + std::to_string(pair_bw.size()) + " p2p transfers]" + + " Aggregate bandwidth: " + total_bw_buff; + + rvs::lp::Log(msg, rvs::logresults); + + result.state = rvs::actionstate::ACTION_RUNNING; + result.status = rvs::actionstatus::ACTION_SUCCESS; + result.output = msg.c_str(); + action_callback(&result); + } + } } return 0; diff --git a/pbqt.so/src/action_run.cpp b/pbqt.so/src/action_run.cpp index 9094314d8..9cdb7d4e8 100644 --- a/pbqt.so/src/action_run.cpp +++ b/pbqt.so/src/action_run.cpp @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -25,7 +25,6 @@ #include "include/action.h" extern "C" { -#include #include } #include @@ -135,19 +134,26 @@ int pbqt_action::run() { return sts; } - if (!prop_test_bandwidth || test_array.size() < 1) { + if (!prop_test_bandwidth) { RVSTRACE_ - // do cleanup - destroy_threads(); + destroy_threads(); + action_result.state = rvs::actionstate::ACTION_COMPLETED; + action_result.status = rvs::actionstatus::ACTION_SUCCESS; + action_result.output = "PBQT Module action " + action_name + " completed"; + action_callback(&action_result); + if (bjson) { rvs::lp::JsonActionEndNodeCreate(); } + return 0; + } - action_result.state = rvs::actionstate::ACTION_COMPLETED; + if (test_array.size() < 1) { + RVSTRACE_ + destroy_threads(); + action_result.state = rvs::actionstate::ACTION_COMPLETED; action_result.status = rvs::actionstatus::ACTION_FAILED; action_result.output = "Parameters not valid. Nothing to execute !!!"; action_callback(&action_result); - if(bjson){ - rvs::lp::JsonActionEndNodeCreate(); - } - return 0; + if (bjson) { rvs::lp::JsonActionEndNodeCreate(); } + return -1; } RVSTRACE_ @@ -167,34 +173,44 @@ int pbqt_action::run() { // start timers if (property_duration) { RVSTRACE_ - timer_final.start(property_duration, true); // ticks only once - } + timer_final.start(property_duration, true); // ticks only once - if (property_log_interval) { - RVSTRACE_ - timer_running.start(property_log_interval); // ticks continuously - } + if (property_log_interval) { + RVSTRACE_ + timer_running.start(property_log_interval); // ticks continuously + } - pbqt_start_time = std::chrono::system_clock::now(); + pbqt_start_time = std::chrono::system_clock::now(); - RVSTRACE_ - do { - if (property_parallel) { - sts = run_parallel(); - } else { - sts = run_single(); - } - pbqt_end_time = std::chrono::system_clock::now(); - uint64_t test_time = time_diff(pbqt_end_time, pbqt_start_time) ; - if(test_time >= property_duration) { - pbqt_action::do_final_average(); - break; - } - } while (brun); + RVSTRACE_ + do { + if (property_parallel) { + sts = run_parallel(); + } else { + sts = run_single(); + } + pbqt_end_time = std::chrono::system_clock::now(); + uint64_t test_time = time_diff(pbqt_end_time, pbqt_start_time); + if((test_time >= property_duration) || ((transfer_method == "transferbench") && (transferbench_test == "alltoall"))) { + pbqt_action::do_final_average(); + break; + } + } while (brun); - RVSTRACE_ - timer_running.stop(); - timer_final.stop(); + RVSTRACE_ + timer_running.stop(); + timer_final.stop(); + + } + else { + + if (property_parallel) { + sts = run_parallel(); + } else { + sts = run_single(); + } + pbqt_action::do_final_average(); + } iter -= step; diff --git a/pbqt.so/src/rvs_module.cpp b/pbqt.so/src/rvs_module.cpp index da94822eb..2a8c26bcd 100644 --- a/pbqt.so/src/rvs_module.cpp +++ b/pbqt.so/src/rvs_module.cpp @@ -24,7 +24,6 @@ *******************************************************************************/ #include "include/rvs_module.h" -#include #include #include "include/rvsloglp.h" @@ -75,7 +74,6 @@ extern "C" int rvs_module_init(void* pMi) { extern "C" int rvs_module_terminate(void) { rvs::hsa::Terminate(); - cleanup_logs(); amdsmi_shut_down(); return 0; } diff --git a/pbqt.so/src/worker.cpp b/pbqt.so/src/worker.cpp index d90f11d28..110e39dec 100644 --- a/pbqt.so/src/worker.cpp +++ b/pbqt.so/src/worker.cpp @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -27,7 +27,6 @@ #ifdef __cplusplus extern "C" { #endif -#include #include #ifdef __cplusplus } @@ -45,13 +44,35 @@ extern "C" { #include "include/gpu_util.h" #include "include/rvsloglp.h" #include "include/rvshsa.h" +#include "TransferBench.hpp" #define MODULE_NAME "PBQT" +/** + * @brief Convert HSA/KFD node index into the HIP device ordinal used by + * TransferBench to index GPU executors and GPU memory. + * @return HIP device index, or -1 if no GPU maps to the given node + */ +static int node_to_hip_index(uint16_t node) { + int num_gpus = TransferBench::GetNumExecutors(TransferBench::EXE_GPU_GFX); + for (int i = 0; i < num_gpus; i++) { + int hip_node = -1; + if (gpu_hip_to_node(i, &hip_node) == 0 && + hip_node == static_cast(node)) { + return i; + } + } + return -1; +} pbqtworker::pbqtworker() { // set to 'true' so that do_transfer() will also work // when parallel: false brun = true; + a2a_mode = 0; + a2a_direct = 1; + a2a_local = 0; + a2a_num_gpus = 0; + use_remote_read = 0; } pbqtworker::~pbqtworker() {} @@ -154,38 +175,218 @@ int pbqtworker::do_transfer() { unsigned int endusec; std::string msg; - uint32_t warm_calls = 1; - uint32_t hot_calls = 1; - bool b2b = false; - msg = "[" + action_name + "] pbqt transfer " + std::to_string(src_node) + " " - + std::to_string(dst_node) + " "; + + std::to_string(dst_node) + " "; rvs::lp::get_ticks(&startsec, &startusec); - if (block_size.size() == 0) { - block_size = pHsa->size_list; - } - for (size_t i = 0; brun && i < block_size.size(); i++) { - current_size = block_size[i]; - sts = pHsa->SendTraffic(src_node, dst_node, current_size, - bidirect, b2b, warm_calls, hot_calls, &duration); - - if (sts) { - msg = "internal error, src: " + std::to_string(src_node) - + " dst: " + std::to_string(dst_node) - + " current size: " + std::to_string(current_size); + if (transfer_method == "transferbench") { + + size_t transfer_block_size = 0; + if(block_size.size() > 0) { + transfer_block_size = block_size[0]; + } + else { + msg = "Transfer block size not set !"; rvs::lp::Err(msg, MODULE_NAME, action_name); - return sts; + return -1; } - { - std::lock_guard lk(cntmutex); - running_size += current_size; - running_duration += duration; + if (transferbench_test == "p2p") { + + // Configure TransferBench parameters + TransferBench::ConfigOptions cfg; + + cfg.general.numIterations = hot_calls; + cfg.general.numWarmups = warm_calls; + + TransferBench::MemType src_mem = TransferBench::MEM_GPU; + TransferBench::MemType dst_mem = TransferBench::MEM_GPU; + + std::vector transfers(bidirect? 2 : 1); + + transfers[0].numBytes = transfer_block_size; + + int src_hip_index = node_to_hip_index(src_node); + int dst_hip_index = node_to_hip_index(dst_node); + if (src_hip_index < 0 || dst_hip_index < 0) { + msg = "no HIP device found for GPU node " + + std::to_string(src_hip_index < 0 ? src_node : dst_node); + rvs::lp::Err(msg, MODULE_NAME, action_name); + return -1; + } + + transfers[0].srcs.push_back({src_mem, + src_mem == TransferBench::MEM_GPU ? src_hip_index : src_node}); + transfers[0].dsts.push_back({dst_mem, + dst_mem == TransferBench::MEM_GPU ? dst_hip_index : dst_node}); + + transfers[0].exeDevice = {executor == "gfx" ? TransferBench::EXE_GPU_GFX : TransferBench::EXE_GPU_DMA, + src_mem == TransferBench::MEM_GPU ? src_hip_index : src_node}; + + transfers[0].exeSubIndex = -1; + + transfers[0].numSubExecs = subexecutor; + + if (bidirect) { + transfers[1].numBytes = transfer_block_size; + + transfers[1].srcs.push_back({dst_mem, + dst_mem == TransferBench::MEM_GPU ? dst_hip_index : dst_node}); + transfers[1].dsts.push_back({src_mem, + src_mem == TransferBench::MEM_GPU ? src_hip_index : src_node}); + + transfers[1].exeDevice = {executor == "gfx" ? TransferBench::EXE_GPU_GFX : TransferBench::EXE_GPU_DMA, + dst_mem == TransferBench::MEM_GPU ? dst_hip_index : dst_node}; + + transfers[1].exeSubIndex = -1; + + transfers[1].numSubExecs = subexecutor; + } + + TransferBench::TestResults results; + + if (!TransferBench::RunTransfers(cfg, transfers, results)) { + for (auto const& err : results.errResults) { + msg = "Transferbench error: " + err.errMsg; + rvs::lp::Err(msg, MODULE_NAME, action_name); + return -1; + } + } + + { + std::lock_guard lk(cntmutex); + + for (size_t t = 0; t < results.tfrResults.size(); t++) { + const auto& res = results.tfrResults[t]; + running_size += results.numTimedIterations * res.numBytes; + running_duration += (res.avgDurationMsec/1000) * results.numTimedIterations; + } + } + + } else if (transferbench_test == "alltoall") { + + int numDetectedGpus = TransferBench::GetNumExecutors(TransferBench::EXE_GPU_GFX); + int numGpus = (a2a_num_gpus > 0 && static_cast(a2a_num_gpus) <= numDetectedGpus) + ? static_cast(a2a_num_gpus) : numDetectedGpus; + + if (numGpus < 2) { + msg += "alltoall requires at least 2 GPUs, detected: " + std::to_string(numDetectedGpus); + rvs::lp::Err(msg, MODULE_NAME, action_name); + return -1; + } + + int numSrcs, numDsts; + switch (a2a_mode) { + case 1: numSrcs = 1; numDsts = 0; break; + case 2: numSrcs = 0; numDsts = 1; break; + default: numSrcs = 1; numDsts = 1; break; + } + + TransferBench::ExeType exeType = (executor == "dma") + ? TransferBench::EXE_GPU_DMA : TransferBench::EXE_GPU_GFX; + TransferBench::MemType memType = TransferBench::MEM_GPU_UNCACHED; + + std::vector transfers; + std::vector> localGpuPairs; + + for (int i = 0; i < numGpus; i++) { + for (int j = 0; j < numGpus; j++) { + + if (i == j) { + if (!a2a_local) continue; + } else if (a2a_direct) { + uint32_t linkType, hopCount; + hipError_t hip_err = hipExtGetLinkTypeAndHopCount(i, j, &linkType, &hopCount); + if (hip_err != hipSuccess || hopCount != 1) continue; + } + + TransferBench::Transfer transfer; + transfer.numBytes = transfer_block_size; + for (int x = 0; x < numSrcs; x++) + transfer.srcs.push_back({memType, i}); + if (numDsts) + transfer.dsts.push_back({memType, j}); + for (int x = 1; x < numDsts; x++) + transfer.dsts.push_back({memType, i}); + + transfer.exeDevice = {exeType, (use_remote_read ? j : i)}; + transfer.exeSubIndex = -1; + transfer.numSubExecs = subexecutor; + + transfers.push_back(transfer); + localGpuPairs.push_back({i, j}); + } + } + + if (transfers.empty()) { + msg += "alltoall: no valid GPU pairs found"; + rvs::lp::Err(msg, MODULE_NAME, action_name); + return -1; + } + + TransferBench::ConfigOptions cfg; + cfg.general.numIterations = hot_calls; + cfg.general.numWarmups = warm_calls; + cfg.gfx.unrollFactor = gfx_unroll; + + TransferBench::TestResults results; + + if (!TransferBench::RunTransfers(cfg, transfers, results)) { + for (auto const& err : results.errResults) { + std::string errmsg = "[" + action_name + "] alltoall error: " + err.errMsg; + rvs::lp::Err(errmsg, MODULE_NAME, action_name); + } + return -1; + } + + gpu_pair_bw.clear(); + for (size_t t = 0; t < results.tfrResults.size(); t++) { + const auto& res = results.tfrResults[t]; + int srcGpu = localGpuPairs[t].first; + int dstGpu = localGpuPairs[t].second; + + int srcNode = -1, dstNode = -1; + gpu_hip_to_node(srcGpu, &srcNode); + gpu_hip_to_node(dstGpu, &dstNode); + + gpu_pair_bw.push_back({srcGpu, dstGpu, srcNode, dstNode, + res.avgBandwidthGbPerSec}); + + } + + } else { + msg += "unknown transferbench_test: " + transferbench_test; + rvs::lp::Err(msg, MODULE_NAME, action_name); + return -1; } } + else { + if (block_size.size() == 0) { + block_size = pHsa->size_list; + } + + for (size_t i = 0; brun && i < block_size.size(); i++) { + current_size = block_size[i]; + sts = pHsa->SendTraffic(src_node, dst_node, current_size, + bidirect, b2b, warm_calls, hot_calls, &duration); + + if (sts) { + msg = "internal error, src: " + std::to_string(src_node) + + " dst: " + std::to_string(dst_node) + + " current size: " + std::to_string(current_size); + rvs::lp::Err(msg, MODULE_NAME, action_name); + return sts; + } + + { + std::lock_guard lk(cntmutex); + running_size += current_size; + running_duration += duration; + } + } + } rvs::lp::get_ticks(&endsec, &endusec); rvs::lp::Log(msg + "start", rvs::logdebug, startsec, startusec); rvs::lp::Log(msg + "finish", rvs::logdebug, endsec, endusec); diff --git a/pbqt.so/src/worker_b2b.cpp b/pbqt.so/src/worker_b2b.cpp index 870bdee2c..485691f00 100644 --- a/pbqt.so/src/worker_b2b.cpp +++ b/pbqt.so/src/worker_b2b.cpp @@ -27,7 +27,6 @@ #ifdef __cplusplus extern "C" { #endif - #include #include #ifdef __cplusplus } diff --git a/pebb.so/CMakeLists.txt b/pebb.so/CMakeLists.txt index 7f624fd24..a904e396e 100644 --- a/pebb.so/CMakeLists.txt +++ b/pebb.so/CMakeLists.txt @@ -1,6 +1,6 @@ ################################################################################ ## -## Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. +## Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. ## ## MIT LICENSE: ## Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -49,7 +49,6 @@ add_definitions(-DLITTLEENDIAN_CPU=1) add_definitions(-DHSA_LARGE_MODEL=) add_definitions(-DHSA_DEPRECATED=) -add_compile_options(-std=c++11) add_compile_options(-pthread) add_compile_options(-Wall ) if (RVS_COVERAGE) @@ -135,11 +134,14 @@ if(NOT EXISTS ${ROCR_LIB_DIR}/${CORE_RUNTIME_LIBRARY}.so) endif() ## define include directories -include_directories(./ ../ pci ${ROCR_INC_DIR} ${HIPRAND_INC_DIR} ${ROCRAND_INC_DIR}) +include_directories(./ ../ pci ${ROCR_INC_DIR} ${HIPRAND_INC_DIR} ${ROCRAND_INC_DIR} ${TRANSFERBENCH_INC_DIR}) +if(TRANSFERBENCH_IBVERBS_INC_DIR) + include_directories(${TRANSFERBENCH_IBVERBS_INC_DIR}) +endif() # Add directories to look for library files to link link_directories(${RVS_LIB_DIR} ${ROCR_LIB_DIR} ${ASAN_LIB_PATH} ${AMD_SMI_LIB_DIR} ${HIPRAND_LIB_DIR} ${ROCRAND_LIB_DIR}) ## additional libraries -set (PROJECT_LINK_LIBS rvslib libpthread.so libpci.so libm.so) +set (PROJECT_LINK_LIBS rvslib libpthread.so ${LIBPCI_TARGET} libm.so) ## define source files set(SOURCES src/rvs_module.cpp src/action.cpp src/action_run.cpp @@ -151,6 +153,9 @@ set_target_properties(${RVS_TARGET} PROPERTIES SUFFIX .so.${LIB_VERSION_STRING} LIBRARY_OUTPUT_DIRECTORY ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}) target_link_libraries(${RVS_TARGET} ${PROJECT_LINK_LIBS} ${CORE_RUNTIME_TARGET}) +if(TRANSFERBENCH_IBVERBS_INC_DIR) + target_link_libraries(${RVS_TARGET} ${CMAKE_DL_LIBS}) +endif() add_dependencies(${RVS_TARGET} rvslib) add_custom_command(TARGET ${RVS_TARGET} POST_BUILD diff --git a/pebb.so/include/action.h b/pebb.so/include/action.h index afcb305fb..55b59752e 100644 --- a/pebb.so/include/action.h +++ b/pebb.so/include/action.h @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -93,13 +93,26 @@ class pebb_action : public rvs::actionbase { int link_type; std::string link_type_string; - //! Number of warm calls (transfer iterations) before bandwidth calculation (hot calls) - //! to ignore few intial transfers for the bandwidth to settle - uint32_t warm_calls; - //! Number of hot calls (transfer iterations) for bandwidth calculation after warm calls - uint32_t hot_calls; - //! set to true for back-to-back transfers (resource allocation only once for entire transfer iterations) - bool b2b; + //! Number of warm calls (transfer iterations) before bandwidth calculation (hot calls) + //! to ignore few intial transfers for the bandwidth to settle + uint32_t warm_calls; + //! Number of hot calls (transfer iterations) for bandwidth calculation after warm calls + uint32_t hot_calls; + //! set to true for back-to-back transfers (resource allocation only once for entire transfer iterations) + bool b2b; + + //! transfer method - TransferBench or Native + std::string transfer_method; + //! transfer executor to use - GPU or SDMA + std::string executor; + //! No. of subexecutors + uint32_t subexecutor; + //! source memory type (e.g., cpu, gpu, null) + std::string source_memory; + //! destination memory type (e.g., cpu, gpu, null) + std::string destination_memory; + //! GFX kernel unroll factor + uint32_t gfx_unroll; protected: int create_threads(); diff --git a/pebb.so/include/worker.h b/pebb.so/include/worker.h index 5a960a8ad..886a4370b 100644 --- a/pebb.so/include/worker.h +++ b/pebb.so/include/worker.h @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -92,6 +92,19 @@ class pebbworker : public rvs::ThreadBase { //! Set b2b void set_b2b(bool _b2b) { b2b = _b2b; } + //! Set transfer method + void set_transfer_method(std::string _transfer_method) { transfer_method = _transfer_method; } + //! Set executor + void set_executor(std::string _executor) { executor = _executor; } + //! Set subexecutor + void set_subexecutor(uint32_t _subexecutor) { subexecutor = _subexecutor; } + //! Set source memory type + void set_source_memory(const std::string& val) { source_memory = val; } + //! Set GFX kernel unroll factor + void set_gfx_unroll(uint32_t val) { gfx_unroll = val; } + //! Set destination memory type + void set_destination_memory(const std::string& val) { destination_memory = val; } + protected: virtual void run(void); @@ -145,6 +158,19 @@ class pebbworker : public rvs::ThreadBase { //! 'true' if back-to-back transfers enabled bool b2b; + //! transfer method - TransferBench or Native + std::string transfer_method; + //! transfer executor to use - GPU or SDMA + std::string executor; + //! No. of subexecutors + uint32_t subexecutor; + //! source memory type (cpu, gpu, null) + std::string source_memory; + //! destination memory type (cpu, gpu, null) + std::string destination_memory; + //! GFX kernel unroll factor + uint32_t gfx_unroll; + //! list of test block sizes std::vector block_size; diff --git a/pebb.so/src/action.cpp b/pebb.so/src/action.cpp index f7a5ce0d2..1b995c46c 100644 --- a/pebb.so/src/action.cpp +++ b/pebb.so/src/action.cpp @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -25,7 +25,6 @@ #include "include/action.h" extern "C" { - #include #include } #include @@ -138,6 +137,48 @@ bool pebb_action::get_all_pebb_config_keys(void) { bsts = false; } + error = property_get(RVS_CONF_TRANSFER_METHOD_KEY, &transfer_method, DEFAULT_TRANSFER_METHOD); + if (error == 1) { + msg = "invalid '" + std::string(RVS_CONF_TRANSFER_METHOD_KEY) + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + error = property_get(RVS_CONF_EXECUTOR_KEY, &executor, DEFAULT_EXECUTOR); + if (error == 1) { + msg = "invalid '" + std::string(RVS_CONF_EXECUTOR_KEY) + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + error = property_get_int(RVS_CONF_SUBEXECUTOR_KEY, &subexecutor, DEFAULT_SUBEXECUTOR); + if (error == 1) { + msg = "invalid '" + std::string(RVS_CONF_SUBEXECUTOR_KEY) + "' key"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + error = property_get("source_memory", &source_memory, DEFAULT_SRC_MEMORY); + if (error == 1) { + msg = "invalid 'source_memory' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + error = property_get("destination_memory", &destination_memory, DEFAULT_DST_MEMORY); + if (error == 1) { + msg = "invalid 'destination_memory' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + error = property_get_int(RVS_CONF_GFX_UNROLL_KEY, &gfx_unroll, DEFAULT_GFX_UNROLL); + if (error == 1) { + msg = "invalid '" + std::string(RVS_CONF_GFX_UNROLL_KEY) + "' key"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + if(!hot_calls) { hot_calls = DEFAULT_HOT_CALLS; } @@ -151,6 +192,14 @@ bool pebb_action::get_all_pebb_config_keys(void) { else if(link_type == 4) link_type_string = "XGMI"; + if(transfer_method == "transferbench") { + if((source_memory == "null") && (destination_memory == "null")) { + msg = "Set proper values for source and destination memory."; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + } + return bsts; } @@ -291,7 +340,7 @@ int pebb_action::create_threads() { p->initialize(srcnode, dstnode, prop_h2d, prop_d2h); } RVSTRACE_ - p->set_name(action_name); + p->set_name(action_name); p->set_stop_name(action_name); p->set_transfer_ix(transfer_ix); p->set_block_sizes(block_size); @@ -299,6 +348,13 @@ int pebb_action::create_threads() { p->set_warm_calls(warm_calls); p->set_b2b(b2b); p->set_loglevel(property_log_level); + p->set_transfer_method(transfer_method); + p->set_executor(executor); + p->set_subexecutor(subexecutor); + p->set_source_memory(source_memory); + p->set_destination_memory(destination_memory); + p->set_gfx_unroll(gfx_unroll); + test_array.push_back(p); } } diff --git a/pebb.so/src/action_run.cpp b/pebb.so/src/action_run.cpp index 958f3444f..f8f77e925 100644 --- a/pebb.so/src/action_run.cpp +++ b/pebb.so/src/action_run.cpp @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -25,7 +25,6 @@ #include "include/action.h" extern "C" { - #include #include } #include @@ -149,35 +148,45 @@ int pebb_action::run() { // start timers if (property_duration) { RVSTRACE_ - timer_final.start(property_duration, true); // ticks only once - } + timer_final.start(property_duration, true); // ticks only once + + if (property_log_interval) { + RVSTRACE_ + timer_running.start(property_log_interval); // ticks continuously + } - if (property_log_interval) { RVSTRACE_ - timer_running.start(property_log_interval); // ticks continuously - } + pebb_start_time = std::chrono::system_clock::now(); + + do { + if (property_parallel) { + sts = run_parallel(); + } else { + sts = run_single(); + } + + pebb_end_time = std::chrono::system_clock::now(); + uint64_t test_time = time_diff(pebb_end_time, pebb_start_time) ; + if(test_time >= property_duration) { + pebb_action::do_final_average(); + break; + } + } while(brun); - RVSTRACE_ - pebb_start_time = std::chrono::system_clock::now(); + RVSTRACE_ + timer_running.stop(); + timer_final.stop(); + } + else { - do { if (property_parallel) { sts = run_parallel(); } else { sts = run_single(); } - pebb_end_time = std::chrono::system_clock::now(); - uint64_t test_time = time_diff(pebb_end_time, pebb_start_time) ; - if(test_time >= property_duration) { - pebb_action::do_final_average(); - break; - } - } while(brun); - - RVSTRACE_ - timer_running.stop(); - timer_final.stop(); + pebb_action::do_final_average(); + } iter -= step; diff --git a/pebb.so/src/rvs_module.cpp b/pebb.so/src/rvs_module.cpp index cf5dc79c0..fbea2adbf 100644 --- a/pebb.so/src/rvs_module.cpp +++ b/pebb.so/src/rvs_module.cpp @@ -24,7 +24,6 @@ *******************************************************************************/ #include "include/rvs_module.h" -#include #include #include @@ -79,7 +78,6 @@ extern "C" int rvs_module_init(void* pMi) { extern "C" int rvs_module_terminate(void) { rvs::lp::Log("[module_terminate] pebb rvs_module_terminate() - entered", rvs::logtrace); - cleanup_logs(); amdsmi_shut_down(); return 0; } diff --git a/pebb.so/src/worker.cpp b/pebb.so/src/worker.cpp index f878aab8b..4ab5c6515 100644 --- a/pebb.so/src/worker.cpp +++ b/pebb.so/src/worker.cpp @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -27,7 +27,6 @@ #ifdef __cplusplus extern "C" { #endif - #include #include #ifdef __cplusplus } @@ -46,12 +45,32 @@ extern "C" { #include "include/rvsloglp.h" #include "include/rvshsa.h" +#include +#include "TransferBench.hpp" + #define MODULE_NAME "PEBB" using std::string; using std::vector; using std::map; +/** + * @brief Convert HSA/KFD node index into the HIP device ordinal used by + * TransferBench to index GPU executors and GPU memory. + * @return HIP device index, or -1 if no GPU maps to the given node + */ +static int node_to_hip_index(uint16_t node) { + int num_gpus = TransferBench::GetNumExecutors(TransferBench::EXE_GPU_GFX); + for (int i = 0; i < num_gpus; i++) { + int hip_node = -1; + if (gpu_hip_to_node(i, &hip_node) == 0 && + hip_node == static_cast(node)) { + return i; + } + } + return -1; +} + extern uint64_t time_diff( std::chrono::time_point t_end, std::chrono::time_point t_start); @@ -156,12 +175,14 @@ int pebbworker::initialize(uint16_t Src, uint16_t Dst, bool h2d, bool d2h) { * * */ int pebbworker::do_transfer() { + double duration; int sts = -1; unsigned int startsec; unsigned int startusec; unsigned int endsec; unsigned int endusec; + std::string msg; RVSTRACE_ @@ -169,45 +190,157 @@ int pebbworker::do_transfer() { if (loglevel >= rvs::logdebug) rvs::lp::get_ticks(&startsec, &startusec); - if (block_size.size() == 0) { - RVSTRACE_ - block_size = pHsa->size_list; - } - - for (size_t i = 0; brun && i < block_size.size(); i++) { - RVSTRACE_ - current_size = block_size[i]; + if (transfer_method == "transferbench") { - if (rvs::lp::Stopping()) { - RVSTRACE_ + size_t transfer_block_size = 0; + if(block_size.size() > 0) { + transfer_block_size = block_size[0]; + } + else { + msg = "Transfer block size not set !"; + rvs::lp::Err(msg, MODULE_NAME, action_name); return -1; } - // Check if unidirectional device(GPU) to host (CPU) - // if so, swap source and destination node + // Configure TransferBench parameters + TransferBench::ConfigOptions cfg; + + cfg.general.numIterations = hot_calls; + cfg.general.numWarmups = warm_calls; + cfg.gfx.unrollFactor = gfx_unroll; + + // Set Memory types + auto str_to_memtype = [](const std::string& mem) -> TransferBench::MemType { + if (mem == "cpu") return TransferBench::MEM_CPU; + if (mem == "gpu") return TransferBench::MEM_GPU; + if (mem == "null") return TransferBench::MEM_NULL; + return TransferBench::MEM_NULL; + }; + + TransferBench::MemType src_mem = str_to_memtype(source_memory); + TransferBench::MemType dst_mem = str_to_memtype(destination_memory); + + std::vector transfers(bidirect? 2 : 1); + + transfers[0].numBytes = transfer_block_size; + + uint16_t _src_node; + uint16_t _dst_node; + if (!prop_h2d && prop_d2h) { - RVSTRACE_ - sts = pHsa->SendTraffic(dst_node, src_node, current_size, - bidirect, b2b, warm_calls, hot_calls, &duration); + _src_node = dst_node; + _dst_node = src_node; } else { - RVSTRACE_ - sts = pHsa->SendTraffic(src_node, dst_node, current_size, - bidirect, b2b, warm_calls, hot_calls, &duration); + _src_node = src_node; + _dst_node = dst_node; } - if (sts) { - std::string msg = "internal error, src: " + std::to_string(src_node) - + " dst: " + std::to_string(dst_node) - + " current size: " + std::to_string(current_size) - + " status "+ std::to_string(sts); + + int gpu_exe_index = node_to_hip_index(dst_node); + if (gpu_exe_index < 0) { + msg = "no HIP device found for GPU node " + std::to_string(dst_node); rvs::lp::Err(msg, MODULE_NAME, action_name); - return sts; + return -1; + } + + if(src_mem != TransferBench::MEM_NULL) { + transfers[0].srcs.push_back({src_mem, + src_mem == TransferBench::MEM_GPU ? node_to_hip_index(_src_node) : _src_node}); + } + + if(dst_mem != TransferBench::MEM_NULL) { + transfers[0].dsts.push_back({dst_mem, + dst_mem == TransferBench::MEM_GPU ? node_to_hip_index(_dst_node) : _dst_node}); + } + + transfers[0].exeDevice = {executor == "gfx" ? TransferBench::EXE_GPU_GFX : TransferBench::EXE_GPU_DMA, + gpu_exe_index}; + + transfers[0].exeSubIndex = -1; + transfers[0].numSubExecs = subexecutor; + + if (bidirect) { + transfers[1].numBytes = transfer_block_size; + + if(dst_mem != TransferBench::MEM_NULL) { + transfers[1].srcs.push_back({dst_mem, + dst_mem == TransferBench::MEM_GPU ? node_to_hip_index(_dst_node) : _dst_node}); + } + + if(src_mem != TransferBench::MEM_NULL) { + transfers[1].dsts.push_back({src_mem, + src_mem == TransferBench::MEM_GPU ? node_to_hip_index(_src_node) : _src_node}); + } + + transfers[1].exeDevice = {executor == "gfx" ? TransferBench::EXE_GPU_GFX : TransferBench::EXE_GPU_DMA, + gpu_exe_index}; + + transfers[1].exeSubIndex = -1; + transfers[1].numSubExecs = subexecutor; } + TransferBench::TestResults results; + + // Initiate TransferBench transfer + if (!TransferBench::RunTransfers(cfg, transfers, results)) { + for (auto const& err : results.errResults) { + msg = "Transferbench error: " + err.errMsg; + rvs::lp::Err(msg, MODULE_NAME, action_name); + return -1; + } + } + + // Update running totals { - RVSTRACE_ std::lock_guard lk(cntmutex); - running_size += current_size * hot_calls; - running_duration += duration; + const auto& res = results.tfrResults[0]; + + running_size += results.numTimedIterations * res.numBytes; + running_duration += (res.avgDurationMsec/1000) * results.numTimedIterations; + } + } + else { + + if (block_size.size() == 0) { + RVSTRACE_ + block_size = pHsa->size_list; + } + + for (size_t i = 0; brun && i < block_size.size(); i++) { + RVSTRACE_ + current_size = block_size[i]; + + if (rvs::lp::Stopping()) { + RVSTRACE_ + return -1; + } + + // Check if unidirectional device(GPU) to host (CPU) + // if so, swap source and destination node + uint32_t num_timed = 0; + if (!prop_h2d && prop_d2h) { + RVSTRACE_ + sts = pHsa->SendTraffic(dst_node, src_node, current_size, + bidirect, b2b, warm_calls, hot_calls, &duration, &num_timed); + } else { + RVSTRACE_ + sts = pHsa->SendTraffic(src_node, dst_node, current_size, + bidirect, b2b, warm_calls, hot_calls, &duration, &num_timed); + } + if (sts) { + std::string msg = "internal error, src: " + std::to_string(src_node) + + " dst: " + std::to_string(dst_node) + + " current size: " + std::to_string(current_size) + + " status "+ std::to_string(sts); + rvs::lp::Err(msg, MODULE_NAME, action_name); + return sts; + } + + { + RVSTRACE_ + std::lock_guard lk(cntmutex); + running_size += current_size * num_timed; + running_duration += duration; + } } } diff --git a/pebb.so/src/worker_b2b.cpp b/pebb.so/src/worker_b2b.cpp index 33979a47c..06baa1a56 100644 --- a/pebb.so/src/worker_b2b.cpp +++ b/pebb.so/src/worker_b2b.cpp @@ -27,7 +27,6 @@ #ifdef __cplusplus extern "C" { #endif - #include #include #ifdef __cplusplus } diff --git a/peqt.so/CMakeLists.txt b/peqt.so/CMakeLists.txt index 354035476..998a99fd6 100644 --- a/peqt.so/CMakeLists.txt +++ b/peqt.so/CMakeLists.txt @@ -110,7 +110,7 @@ include_directories(./ ../ ${HSA_INC_DIR}) # Add directories to look for library files to link link_directories(${RVS_LIB_DIR} ${ASAN_LIB_PATH} ${HSA_LIB_DIR} ${ASAN_LIB_PATH} ${AMD_SMI_LIB_DIR} ${HIPRAND_LIB_DIR} ${ROCRAND_LIB_DIR}) ## additional libraries -set (PROJECT_LINK_LIBS rvslib libpci.so libm.so) +set (PROJECT_LINK_LIBS rvslib ${LIBPCI_TARGET} libm.so) ## define source files set(SOURCES src/rvs_module.cpp src/action.cpp) diff --git a/peqt.so/src/action.cpp b/peqt.so/src/action.cpp index 34c7cff23..2a43f1444 100644 --- a/peqt.so/src/action.cpp +++ b/peqt.so/src/action.cpp @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -201,7 +201,7 @@ bool peqt_action::get_gpu_all_pcie_capabilities(struct pci_dev *dev, if (bjson){ json_pcaps_node = json_node_create(MODULE_NAME, action_name.c_str(), rvs::logresults); - + if (json_pcaps_node != NULL) { rvs::lp::AddString(json_pcaps_node, RVS_JSON_LOG_GPU_ID_KEY, std::to_string(gpu_id)); @@ -215,7 +215,7 @@ bool peqt_action::get_gpu_all_pcie_capabilities(struct pci_dev *dev, string prop_name = it->first.substr(it->first.find_last_of(".") + 1); bool prop_found = false; for (i = 0; i < PCI_DEV_NUM_CAPABILITIES; i++) { - if ((prop_name == pcie_cap_names[i]) && + if ((prop_name == pcie_cap_names[i]) && ( dev != NULL )){ prop_found = true; // call the capability's corresponding function @@ -264,6 +264,9 @@ bool peqt_action::get_gpu_all_pcie_capabilities(struct pci_dev *dev, map::iterator it_pb_pm_state = pb_op_pm_states_encodings_map.find (prop_name.substr(0, pos_pb_pm_state)); + if (it_pb_pm_state == pb_op_pm_states_encodings_map.end()) { + continue; + } uint8_t pb_op_pm_state = it_pb_pm_state->second; std::size_t pos_pb_type = @@ -273,11 +276,17 @@ bool peqt_action::get_gpu_all_pcie_capabilities(struct pci_dev *dev, pb_op_pm_types_encodings_map.find (prop_name.substr(pos_pb_pm_state + 1, pos_pb_type - pos_pb_pm_state - 1)); + if (it_pb_type == pb_op_pm_types_encodings_map.end()) { + continue; + } uint8_t pb_op_pm_type = it_pb_type->second; map::iterator it_pb_power_rail = pb_op_pm_power_rails_encodings_map.find (prop_name.substr(pos_pb_type + 1)); + if (it_pb_power_rail == pb_op_pm_power_rails_encodings_map.end()) { + continue; + } uint8_t pb_op_power_rail = it_pb_power_rail->second; // query for power budgeting capabilities get_pwr_budgeting(dev, pb_op_pm_state, pb_op_pm_type, @@ -358,7 +367,7 @@ int peqt_action::run(void) { return -1; } if (bjson){ - json_add_primary_fields(std::string(MODULE_NAME), action_name); + json_add_primary_fields(std::string(MODULE_NAME), action_name); } // get the pci_access structure pacc = pci_alloc(); @@ -426,7 +435,7 @@ int peqt_action::run(void) { action_result.output = msg; action_callback(&action_result); - return false; + return -1; } // fill in the info @@ -437,11 +446,12 @@ int peqt_action::run(void) { // computes the actual dev's location_id (sysfs entry) uint16_t dev_location_id = ((((uint16_t) (dev->bus)) << 8) | ((uint16_t) (dev->dev)) << 3 | ((uint16_t) (dev->func)) ); + uint16_t dev_domain = static_cast(dev->domain); // check if this pci_dev corresponds to one of the AMD GPUs uint16_t gpu_id; - // if not and AMD GPU just continue - if (rvs::gpulist::location2gpu(dev_location_id, &gpu_id)) { + // if not an AMD GPU just continue + if (rvs::gpulist::domlocation2gpu(dev_domain, dev_location_id, &gpu_id)) { RVSTRACE_ continue; } @@ -508,16 +518,16 @@ int peqt_action::run(void) { rvs::lp::Log(msg, rvs::logresults); if (bjson) { - rvs::lp::JsonActionEndNodeCreate(); + rvs::lp::JsonActionEndNodeCreate(); } action_result.state = rvs::actionstate::ACTION_COMPLETED; - action_result.status = rvs::actionstatus::ACTION_SUCCESS; + action_result.status = pci_infra_qual_result + ? rvs::actionstatus::ACTION_SUCCESS + : rvs::actionstatus::ACTION_FAILED; action_result.output = "PEQT Module action " + action_name + " completed"; action_callback(&action_result); RVSTRACE_ - return 0; + return pci_infra_qual_result ? 0 : -1; } - - diff --git a/peqt.so/src/rvs_module.cpp b/peqt.so/src/rvs_module.cpp index 1a393168b..f4ae5d506 100644 --- a/peqt.so/src/rvs_module.cpp +++ b/peqt.so/src/rvs_module.cpp @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2025 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -67,7 +67,6 @@ extern "C" int rvs_module_init(void* pMi) { } extern "C" int rvs_module_terminate(void) { - cleanup_logs(); amdsmi_shut_down(); return 0; } diff --git a/perf.so/include/action.h b/perf.so/include/action.h deleted file mode 100644 index 37570b8ae..000000000 --- a/perf.so/include/action.h +++ /dev/null @@ -1,121 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#ifndef PERF_SO_INCLUDE_ACTION_H_ -#define PERF_SO_INCLUDE_ACTION_H_ - -#ifdef __cplusplus -extern "C" { -#endif -#include -#ifdef __cplusplus -} -#endif - -#include -#include -#include - -#include "include/rvsactionbase.h" - -using std::vector; -using std::string; -using std::map; - -/** - * @class perf_action - * @ingroup PERF - * - * @brief PERF action implementation class - * - * Derives from rvs::actionbase and implements actual action functionality - * in its run() method. - * - */ -class perf_action: public rvs::actionbase { - public: - perf_action(); - virtual ~perf_action(); - - virtual int run(void); - - std::string perf_ops_type; - - protected: - - //! stress test ramp duration - uint64_t perf_ramp_interval; - //! maximum allowed number of target_stress violations - int perf_max_violations; - //! specifies whether to copy the matrices to the GPU before each - //! SGEMM operation - bool perf_copy_matrix; - //! target stress (in GFlops) that the GPU will try to achieve - float perf_target_stress; - //! GFlops tolerance (how much the GFlops can fluctuare after - //! the ramp period for the test to succeed) - float perf_tolerance; - - //Alpha and beta value - float perf_alpha_val; - float perf_beta_val; - - //! matrix size for SGEMM - uint64_t perf_matrix_size_a; - uint64_t perf_matrix_size_b; - uint64_t perf_matrix_size_c; - - //! Parameter to heat up - uint64_t perf_hot_calls; - - //! Tranpose set to none or enabled - int perf_trans_a; - int perf_trans_b; - - //! Leading offset values - int perf_lda_offset; - int perf_ldb_offset; - int perf_ldc_offset; - int perf_ldd_offset; - - friend class PERFWorker; - - /** - * @brief reads all perf configuration keys from - * the module's properties collection - * @return true if no fatal error occured, false otherwise - */ - bool get_all_perf_config_keys(void); - - - /** - * @brief gets the number of ROCm compatible AMD GPUs - * @return run number of GPUs - */ - int get_num_amd_gpu_devices(void); - int get_all_selected_gpus(void); - bool do_gpu_stress_test(map perf_gpus_device_index); -}; - -#endif // PERF_SO_INCLUDE_ACTION_H_ diff --git a/perf.so/include/perf_worker.h b/perf.so/include/perf_worker.h deleted file mode 100644 index 4599c65d5..000000000 --- a/perf.so/include/perf_worker.h +++ /dev/null @@ -1,273 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#ifndef PERF_SO_INCLUDE_PERF_WORKER_H_ -#define PERF_SO_INCLUDE_PERF_WORKER_H_ - -#include -#include -#include "include/rvsthreadbase.h" -#include "include/rvs_blas.h" -#include "include/rvsactionbase.h" -#include "include/action.h" - -#define PERF_RESULT_PASS_MESSAGE "true" -#define PERF_RESULT_FAIL_MESSAGE "false" - -/** - * @class PERFWorker - * @ingroup PERF - * - * @brief PERFWorker action implementation class - * - * Derives from rvs::ThreadBase and implements actual action functionality - * in its run() method. - * - */ -class PERFWorker : public rvs::ThreadBase { - public: - PERFWorker(); - virtual ~PERFWorker(); - - //! sets action name - void set_name(const std::string& name) { action_name = name; } - //! sets action - void set_action(const perf_action& _action) { action = _action; } - //! returns action name - const std::string& get_name(void) { return action_name; } - //! sets GPU ID - void set_gpu_id(uint16_t _gpu_id) { gpu_id = _gpu_id; } - //! returns GPU ID - uint16_t get_gpu_id(void) { return gpu_id; } - - //! sets the GPU index - void set_gpu_device_index(int _gpu_device_index) { - gpu_device_index = _gpu_device_index; - } - //! returns the GPU index - int get_gpu_device_index(void) { return gpu_device_index; } - - //! sets the run delay - void set_run_wait_ms(uint64_t _run_wait_ms) { run_wait_ms = _run_wait_ms; } - //! returns the run delay - uint64_t get_run_wait_ms(void) { return run_wait_ms; } - - //! sets the total stress test run duration - void set_run_duration_ms(uint64_t _run_duration_ms) { - run_duration_ms = _run_duration_ms; - } - //! returns the total stress test run duration - uint64_t get_run_duration_ms(void) { return run_duration_ms; } - - //! sets the stress test ramp duration - void set_ramp_interval(uint64_t _ramp_interval) { - ramp_interval = _ramp_interval; - } - //! returns the stress test ramp duration - uint64_t get_ramp_interval(void) { return ramp_interval; } - - //! sets the time interval at which the module reports the average GFlops - void set_log_interval(uint64_t _log_interval) { - log_interval = _log_interval; - } - //! returns the time interval at which the module reports the average GFlops - uint64_t get_log_interval(void) { return log_interval; } - - //! sets the maximum allowed number of target_stress violations - void set_max_violations(uint64_t _max_violations) { - max_violations = _max_violations; - } - //! returns the maximum allowed number of target_stress violations - uint64_t get_max_violations(void) { return max_violations; } - - //! sets the copy_matrix (true = the matrix will be copied to GPU each - //! time a new SGEMM will run, false = the matrix will be copied only once) - void set_copy_matrix(bool _copy_matrix) { copy_matrix = _copy_matrix; } - //! returns the copy_matrix value - bool get_copy_matrix(void) { return copy_matrix; } - - //! sets the target stress (in GFlops) that the GPU will try to achieve - void set_target_stress(float _target_stress) { - target_stress = _target_stress; - } - //! returns the target stress (in GFlops) that the GPU will try to achieve - float get_target_stress(void) { return target_stress; } - - //! sets hot calls - void set_perf_hot_calls(uint64_t _hot_calls) { - perf_hot_calls = _hot_calls; - } - - //! sets hot calls - uint64_t get_perf_hot_calls(void) { - return perf_hot_calls; - } - - //! sets the SGEMM matrix size - void set_matrix_size_a(uint64_t _matrix_size_a) { - matrix_size_a = _matrix_size_a; - } - //! sets the SGEMM matrix size - void set_matrix_size_b(uint64_t _matrix_size_b) { - matrix_size_b = _matrix_size_b; - } - //! sets the SGEMM matrix size - void set_matrix_size_c(uint64_t _matrix_size_c) { - matrix_size_c = _matrix_size_c; - } - //! sets the transpose matrix a - void set_matrix_transpose_a(int transa) { - perf_trans_a = transa; - } - //! sets the transpose matrix b - void set_matrix_transpose_b(int transb) { - perf_trans_b = transb; - } - //! sets alpha val - void set_alpha_val(float alpha_val) { - perf_alpha_val = alpha_val; - } - //! sets beta val - void set_beta_val(float beta_val) { - perf_beta_val = beta_val; - } - - //! sets offsets - void set_lda_offset(int lda) { - perf_lda_offset = lda; - } - //! sets offsets - void set_ldb_offset(int ldb) { - perf_ldb_offset = ldb; - } - //! sets offsets - void set_ldc_offset(int ldc) { - perf_ldc_offset = ldc; - } - //! sets offsets - void set_ldd_offset(int ldd) { - perf_ldd_offset = ldd; - } - - //! returns the SGEMM matrix size - uint64_t get_matrix_size_a(void) { return matrix_size_a; } - - //! returns the SGEMM matrix size - uint64_t get_matrix_size_b(void) { return matrix_size_b; } - - //! returns the SGEMM matrix size - uint64_t get_matrix_size_c(void) { return matrix_size_c; } - - //! sets the GFlops tolerance - void set_tolerance(float _tolerance) { tolerance = _tolerance; } - //! returns the GFlops tolerance - float get_tolerance(void) { return tolerance; } - - - //! returns the difference (in milliseconds) between 2 points in time - uint64_t time_diff( - std::chrono::time_point t_end, - std::chrono::time_point t_start); - - //! sets the JSON flag - static void set_use_json(bool _bjson) { bjson = _bjson; } - //! returns the JSON flag - static bool get_use_json(void) { return bjson; } - - void set_perf_ops_type(std::string _ops_type) { perf_ops_type = _ops_type; } - - protected: - void setup_blas(int *error, std::string *err_description); - void hit_max_gflops(int *error, std::string *err_description); - bool do_perf_ramp(int *error, std::string *err_description); - bool do_perf_stress_test(int *error, std::string *err_description); - void log_perf_test_result(bool perf_test_passed); - virtual void run(void); - void log_to_json(const std::string &key, const std::string &value, - int log_level); - void log_interval_gflops(double gflops_interval); - bool check_gflops_violation(double gflops_interval); - void check_target_stress(double gflops_interval); - void usleep_ex(uint64_t microseconds); - - protected: - //! name of the action - std::string action_name; - //! action instance - perf_action action; - //! index of the GPU that will run the stress test - int gpu_device_index; - //Matrix transpose A - int perf_trans_a; - //Matrix transpose B - int perf_trans_b; - //! ID of the GPU that will run the stress test - uint16_t gpu_id; - //PERF aplha value - float perf_alpha_val; - //PERF beta value - float perf_beta_val; - //leading offsets - int perf_lda_offset; - int perf_ldb_offset; - int perf_ldc_offset; - int perf_ldd_offset; - //! stress test run delay - uint64_t run_wait_ms; - //! stress test run duration - uint64_t run_duration_ms; - //! stress test ramp duration - uint64_t ramp_interval; - //! time interval at which the module reports the average GFlops - uint64_t log_interval; - //! maximum allowed number of target_stress violations - uint64_t max_violations; - //! specifies whether to copy the matrix to the GPU for each SGEMM operation - bool copy_matrix; - //! target stress (in GFlops) that the GPU will try to achieve - float target_stress; - //! GFlops tolerance (how much the GFlops can fluctuare after - //! the ramp period for the test to succeed) - float tolerance; - //! SGEMM matrix size - uint64_t matrix_size_a; - uint64_t matrix_size_b; - uint64_t matrix_size_c; - //num of hot calls - uint64_t perf_hot_calls; - //! actual ramp time in case the GPU achieves the given target_stress Gflops - uint64_t ramp_actual_time; - //! rvs_blas pointer - std::unique_ptr gpu_blas; - //! max gflops achieved during the stress test - double max_gflops; - //! delay used to reduce SGEMM frequency - double delay_target_stress; - //! TRUE if JSON output is required - static bool bjson; - //Type of operation - std::string perf_ops_type; -}; - -#endif // PERF_SO_INCLUDE_PERF_WORKER_H_ diff --git a/perf.so/include/rvs_module.h b/perf.so/include/rvs_module.h deleted file mode 100644 index 73d65167c..000000000 --- a/perf.so/include/rvs_module.h +++ /dev/null @@ -1,31 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#ifndef PERF_SO_INCLUDE_RVS_MODULE_H_ -#define PERF_SO_INCLUDE_RVS_MODULE_H_ - -#include "include/rvsliblog.h" - - -#endif // PERF_SO_INCLUDE_RVS_MODULE_H_ diff --git a/perf.so/src/.gitignore b/perf.so/src/.gitignore deleted file mode 100644 index 6677c8735..000000000 --- a/perf.so/src/.gitignore +++ /dev/null @@ -1 +0,0 @@ -/libmain.cpp diff --git a/perf.so/src/action.cpp b/perf.so/src/action.cpp deleted file mode 100644 index 1f3e6890a..000000000 --- a/perf.so/src/action.cpp +++ /dev/null @@ -1,491 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#include "include/action.h" - -#include -#include -#include -#include -#include -#include -#include - -#define __HIP_PLATFORM_HCC__ -#include "hip/hip_runtime.h" -#include "hip/hip_runtime_api.h" - -#include "include/rvs_key_def.h" -#include "include/perf_worker.h" -#include "include/gpu_util.h" -#include "include/rvs_util.h" -#include "include/rvsactionbase.h" -#include "include/rvsloglp.h" - -using std::string; -using std::vector; -using std::map; -using std::regex; - -#define RVS_CONF_RAMP_INTERVAL_KEY "ramp_interval" -#define RVS_CONF_LOG_INTERVAL_KEY "log_interval" -#define RVS_CONF_MAX_VIOLATIONS_KEY "max_violations" -#define RVS_CONF_COPY_MATRIX_KEY "copy_matrix" -#define RVS_CONF_TARGET_STRESS_KEY "target_stress" -#define RVS_CONF_TOLERANCE_KEY "tolerance" -#define RVS_CONF_HOT_CALLS "hot_calls" -#define RVS_CONF_MATRIX_SIZE_KEYA "matrix_size_a" -#define RVS_CONF_MATRIX_SIZE_KEYB "matrix_size_b" -#define RVS_CONF_MATRIX_SIZE_KEYC "matrix_size_c" -#define RVS_CONF_PERF_OPS_TYPE "ops_type" -#define RVS_CONF_TRANS_A "transa" -#define RVS_CONF_TRANS_B "transb" -#define RVS_CONF_ALPHA_VAL "alpha" -#define RVS_CONF_BETA_VAL "beta" -#define RVS_CONF_LDA_OFFSET "lda" -#define RVS_CONF_LDB_OFFSET "ldb" -#define RVS_CONF_LDC_OFFSET "ldc" -#define RVS_CONF_LDD_OFFSET "ldd" - - -#define PERF_DEFAULT_RAMP_INTERVAL 5000 -#define PERF_DEFAULT_LOG_INTERVAL 1000 -#define PERF_DEFAULT_MAX_VIOLATIONS 0 -#define PERF_DEFAULT_TOLERANCE 0.1 -#define PERF_DEFAULT_COPY_MATRIX true -#define PERF_DEFAULT_MATRIX_SIZE 5760 -#define PERF_DEFAULT_HOT_CALLS 0 -#define PERF_DEFAULT_TRANS_A 0 -#define PERF_DEFAULT_TRANS_B 1 -#define PERF_DEFAULT_ALPHA_VAL 1 -#define PERF_DEFAULT_BETA_VAL 1 -#define PERF_DEFAULT_LDA_OFFSET 0 -#define PERF_DEFAULT_LDB_OFFSET 0 -#define PERF_DEFAULT_LDC_OFFSET 0 -#define PERF_DEFAULT_LDD_OFFSET 0 - -#define RVS_DEFAULT_PARALLEL false -#define RVS_DEFAULT_DURATION 0 - -#define PERF_NO_COMPATIBLE_GPUS "No AMD compatible GPU found!" - -#define FLOATING_POINT_REGEX "^[0-9]*\\.?[0-9]+$" - -#define JSON_CREATE_NODE_ERROR "JSON cannot create node" -#define PERF_DEFAULT_OPS_TYPE "sgemm" - - -static constexpr auto MODULE_NAME = "perf"; -static constexpr auto MODULE_NAME_CAPS = "PERF"; -/** - * @brief default class constructor - */ -perf_action::perf_action() { - module_name = MODULE_NAME; - bjson = false; -} - -/** - * @brief class destructor - */ -perf_action::~perf_action() { - property.clear(); -} - -/** - * @brief runs the PERF test stress session - * @param perf_gpus_device_index map - * @return true if no error occured, false otherwise - */ -bool perf_action::do_gpu_stress_test(map perf_gpus_device_index) { - size_t k = 0; - for (;;) { - unsigned int i = 0; - if (property_wait != 0) // delay perf execution - sleep(property_wait); - - vector workers(perf_gpus_device_index.size()); - - map::iterator it; - - // all worker instances have the same json settings - PERFWorker::set_use_json(bjson); - - for (it = perf_gpus_device_index.begin(); - it != perf_gpus_device_index.end(); ++it) { - // set worker thread stress test params - workers[i].set_name(action_name); - workers[i].set_action(*this); - workers[i].set_gpu_id(it->second); - workers[i].set_gpu_device_index(it->first); - workers[i].set_run_wait_ms(property_wait); - workers[i].set_run_duration_ms(property_duration); - workers[i].set_ramp_interval(perf_ramp_interval); - workers[i].set_log_interval(property_log_interval); - workers[i].set_max_violations(perf_max_violations); - workers[i].set_copy_matrix(perf_copy_matrix); - workers[i].set_target_stress(perf_target_stress); - workers[i].set_tolerance(perf_tolerance); - workers[i].set_perf_hot_calls(perf_hot_calls); - workers[i].set_matrix_size_a(perf_matrix_size_a); - workers[i].set_matrix_size_b(perf_matrix_size_b); - workers[i].set_matrix_size_c(perf_matrix_size_c); - workers[i].set_perf_ops_type(perf_ops_type); - workers[i].set_matrix_transpose_a(perf_trans_a); - workers[i].set_matrix_transpose_b(perf_trans_b); - workers[i].set_alpha_val(perf_alpha_val); - workers[i].set_beta_val(perf_beta_val); - workers[i].set_lda_offset(perf_lda_offset); - workers[i].set_ldb_offset(perf_ldb_offset); - workers[i].set_ldc_offset(perf_ldc_offset); - workers[i].set_ldd_offset(perf_ldd_offset); - - i++; - } - - if (property_parallel) { - for (i = 0; i < perf_gpus_device_index.size(); i++) - workers[i].start(); - - // join threads - for (i = 0; i < perf_gpus_device_index.size(); i++) - workers[i].join(); - } else { - for (i = 0; i < perf_gpus_device_index.size(); i++) { - workers[i].start(); - workers[i].join(); - - // check if stop signal was received - if (rvs::lp::Stopping()) - return false; - } - } - - // check if stop signal was received - if (rvs::lp::Stopping()) - return false; - - if (property_count != 0) { - k++; - if (k == property_count) - break; - } - } - - return rvs::lp::Stopping() ? false : true; -} - -/** - * @brief reads all PERF-related configuration keys from - * the module's properties collection - * @return true if no fatal error occured, false otherwise - */ -bool perf_action::get_all_perf_config_keys(void) { - int error; - string msg, ststress; - bool bsts = true; - - if ((error = - property_get(RVS_CONF_TARGET_STRESS_KEY, &perf_target_stress))) { - switch (error) { // is mandatory => PERF cannot continue - case 1: - msg = "invalid '" + std::string(RVS_CONF_TARGET_STRESS_KEY) + - "' key value " + ststress; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - break; - - case 2: - msg = "key '" + std::string(RVS_CONF_TARGET_STRESS_KEY) + - "' was not found"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - } - bsts = false; - } - - if (property_get_int(RVS_CONF_RAMP_INTERVAL_KEY, - &perf_ramp_interval, PERF_DEFAULT_RAMP_INTERVAL)) { - msg = "invalid '" + - std::string(RVS_CONF_RAMP_INTERVAL_KEY) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - if (property_get_int(RVS_CONF_LOG_INTERVAL_KEY, - &property_log_interval, PERF_DEFAULT_LOG_INTERVAL)) { - msg = "invalid '" + - std::string(RVS_CONF_LOG_INTERVAL_KEY) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - if (property_get_int(RVS_CONF_MAX_VIOLATIONS_KEY, &perf_max_violations, - PERF_DEFAULT_MAX_VIOLATIONS)) { - msg = "invalid '" + - std::string(RVS_CONF_MAX_VIOLATIONS_KEY) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - if (property_get(RVS_CONF_COPY_MATRIX_KEY, &perf_copy_matrix, - PERF_DEFAULT_COPY_MATRIX)) { - msg = "invalid '" + - std::string(RVS_CONF_COPY_MATRIX_KEY) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - if (property_get(RVS_CONF_TOLERANCE_KEY, &perf_tolerance, - PERF_DEFAULT_TOLERANCE)) { - msg = "invalid '" + - std::string(RVS_CONF_TOLERANCE_KEY) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - if (property_get(RVS_CONF_PERF_OPS_TYPE, &perf_ops_type, - PERF_DEFAULT_OPS_TYPE)) { - msg = "invalid '" + - std::string(RVS_CONF_PERF_OPS_TYPE) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - error = property_get_int(RVS_CONF_HOT_CALLS, &perf_hot_calls, PERF_DEFAULT_HOT_CALLS); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_HOT_CALLS) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - - error = property_get_int(RVS_CONF_MATRIX_SIZE_KEYA, &perf_matrix_size_a, PERF_DEFAULT_MATRIX_SIZE); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_MATRIX_SIZE_KEYA) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - error = property_get_int(RVS_CONF_MATRIX_SIZE_KEYB, &perf_matrix_size_b, PERF_DEFAULT_MATRIX_SIZE); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_MATRIX_SIZE_KEYB) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - error = property_get_int(RVS_CONF_MATRIX_SIZE_KEYC, &perf_matrix_size_c, PERF_DEFAULT_MATRIX_SIZE); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_MATRIX_SIZE_KEYC) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - error = property_get_int(RVS_CONF_TRANS_A, &perf_trans_a, PERF_DEFAULT_TRANS_A); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_TRANS_A) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - error = property_get_int(RVS_CONF_TRANS_B, &perf_trans_b, PERF_DEFAULT_TRANS_B); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_TRANS_B) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - error = property_get(RVS_CONF_ALPHA_VAL, &perf_alpha_val, PERF_DEFAULT_ALPHA_VAL); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_ALPHA_VAL) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - error = property_get(RVS_CONF_BETA_VAL, &perf_beta_val, PERF_DEFAULT_BETA_VAL); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_BETA_VAL) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - error = property_get_int(RVS_CONF_LDA_OFFSET, &perf_lda_offset, PERF_DEFAULT_LDA_OFFSET); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_LDA_OFFSET) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - error = property_get_int(RVS_CONF_LDB_OFFSET, &perf_ldb_offset, PERF_DEFAULT_LDB_OFFSET); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_LDB_OFFSET) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - error = property_get_int(RVS_CONF_LDC_OFFSET, &perf_ldc_offset, PERF_DEFAULT_LDC_OFFSET); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_LDC_OFFSET) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - error = property_get_int(RVS_CONF_LDD_OFFSET, &perf_ldd_offset, PERF_DEFAULT_LDD_OFFSET); - if (error == 1) { - msg = "invalid '" + - std::string(RVS_CONF_LDD_OFFSET) + "' key value"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - bsts = false; - } - - return bsts; -} - - -/** - * @brief gets the number of ROCm compatible AMD GPUs - * @return run number of GPUs - */ -int perf_action::get_num_amd_gpu_devices(void) { - int hip_num_gpu_devices; - string msg; - - hipGetDeviceCount(&hip_num_gpu_devices); - if (hip_num_gpu_devices == 0) { // no AMD compatible GPU - msg = action_name + " " + MODULE_NAME + " " + PERF_NO_COMPATIBLE_GPUS; - rvs::lp::Log(msg, rvs::logerror); - - if (bjson) { - unsigned int sec; - unsigned int usec; - rvs::lp::get_ticks(&sec, &usec); - void *json_root_node = rvs::lp::LogRecordCreate(MODULE_NAME, - action_name.c_str(), rvs::loginfo, sec, usec); - if (!json_root_node) { - // log the error - string msg = std::string(JSON_CREATE_NODE_ERROR); - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - return -1; - } - - rvs::lp::AddString(json_root_node, "ERROR", PERF_NO_COMPATIBLE_GPUS); - rvs::lp::LogRecordFlush(json_root_node); - } - return 0; - } - return hip_num_gpu_devices; -} - -/** - * @brief gets all selected GPUs and starts the worker threads - * @return run result - */ -int perf_action::get_all_selected_gpus(void) { - int hip_num_gpu_devices; - bool amd_gpus_found = false; - map perf_gpus_device_index; - std::string msg; - - hip_num_gpu_devices = get_num_amd_gpu_devices(); - if (hip_num_gpu_devices < 1) - return hip_num_gpu_devices; - - // iterate over all available & compatible AMD GPUs - amd_gpus_found = fetch_gpu_list(hip_num_gpu_devices, perf_gpus_device_index, - property_device, property_device_id, property_device_all, - property_device_index, property_device_index_all); - if (amd_gpus_found) { - if (do_gpu_stress_test(perf_gpus_device_index)) - return 0; - - return -1; - } else { - msg = "No devices match criteria from the test configuration."; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - return -1; - } - - return 0; -} - -/** - * @brief runs the whole PERF logic - * @return run result - */ -int perf_action::run(void) { - string msg; - rvs::action_result_t action_result; - - - - if (!get_all_common_config_keys()) { - - action_result.state = rvs::actionstate::ACTION_COMPLETED; - action_result.status = rvs::actionstatus::ACTION_FAILED; - action_result.output = "Error in common configuration keys."; - action_callback(&action_result); - return -1; - } - - if (!get_all_perf_config_keys()) { - - action_result.state = rvs::actionstate::ACTION_COMPLETED; - action_result.status = rvs::actionstatus::ACTION_FAILED; - action_result.output = "Error in PERF configuration keys."; - action_callback(&action_result); - return -1; - } - - if (property_duration > 0 && (property_duration < perf_ramp_interval)) { - msg = "'" + - std::string(RVS_CONF_DURATION_KEY) + "' cannot be less than '" + - std::string(RVS_CONF_RAMP_INTERVAL_KEY) + "'"; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - - action_result.state = rvs::actionstate::ACTION_COMPLETED; - action_result.status = rvs::actionstatus::ACTION_FAILED; - action_result.output = "Error in common configuration keys."; - action_callback(&action_result); - return -1; - } - - auto res = get_all_selected_gpus(); - - action_result.state = rvs::actionstate::ACTION_COMPLETED; - action_result.status = (!res) ? rvs::actionstatus::ACTION_SUCCESS : rvs::actionstatus::ACTION_FAILED; - action_result.output = "PERF Module action " + action_name + " completed"; - action_callback(&action_result); - - return true; -} - diff --git a/perf.so/src/perf_worker.cpp b/perf.so/src/perf_worker.cpp deleted file mode 100644 index ccfade8ee..000000000 --- a/perf.so/src/perf_worker.cpp +++ /dev/null @@ -1,303 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2025 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#include "include/perf_worker.h" - -#include -#include -#include -#include - -#include "include/rvs_blas.h" -#include "include/rvs_module.h" -#include "include/rvsloglp.h" - -#define MODULE_NAME "perf" - -#define PERF_MEM_ALLOC_ERROR "memory allocation error!" -#define PERF_BLAS_ERROR "memory/blas error!" -#define PERF_BLAS_MEMCPY_ERROR "HostToDevice mem copy error!" - -#define PERF_MAX_GFLOPS_OUTPUT_KEY "Gflop" -#define PERF_FLOPS_PER_OP_OUTPUT_KEY "flops_per_op" -#define PERF_BYTES_COPIED_PER_OP_OUTPUT_KEY "bytes_copied_per_op" -#define PERF_TRY_OPS_PER_SEC_OUTPUT_KEY "try_ops_per_sec" - -#define PERF_LOG_GFLOPS_INTERVAL_KEY "Gflops" -#define PERF_JSON_LOG_GPU_ID_KEY "gpu_id" - -#define PROC_DEC_INC_SGEMM_FREQ_DELAY 10 - -#define NMAX_MS_GPU_RUN_PEAK_PERFORMANCE 1000 -#define NMAX_MS_SGEMM_OPS_RAMP_SUB_INTERVAL 1000 -#define USLEEP_MAX_VAL (1000000 - 1) - -#define PERF_COPY_MATRIX_MSG "copy matrix" -#define PERF_START_MSG "start" -#define PERF_PASS_KEY "pass" -#define PERF_RAMP_EXCEEDED_MSG "ramp time exceeded" -#define PERF_TARGET_ACHIEVED_MSG "target achieved" -#define PERF_STRESS_VIOLATION_MSG "stress violation" - -using std::string; - -bool PERFWorker::bjson = false; - -PERFWorker::PERFWorker() {} -PERFWorker::~PERFWorker() {} - -/** - * @brief performs the rvsBlas setup - * @param error pointer to a memory location where the error code will be stored - * @param err_description stores the error description if any - */ -void PERFWorker::setup_blas(int *error, string *err_description) { - - std::string blas_source = "rocblas"; - std::string compute_type = "fp32_r"; - - *error = 0; - // setup rvsBlas - gpu_blas = std::unique_ptr( - new rvs_blas(gpu_device_index, matrix_size_a, matrix_size_b, matrix_size_c, - "default", perf_trans_a, perf_trans_b, - perf_alpha_val, perf_beta_val, - perf_lda_offset, perf_ldb_offset, perf_ldc_offset, perf_ldd_offset, perf_ops_type, - "", "", 0, 0, 0, 0, 0, blas_source, compute_type, "", "", "", 0, perf_hot_calls)); - - if (!gpu_blas) { - *error = 1; - *err_description = PERF_MEM_ALLOC_ERROR; - return; - } - - if (gpu_blas->error()) { - *error = 1; - *err_description = PERF_MEM_ALLOC_ERROR; - return; - } - - // generate random matrix & copy it to the GPU - gpu_blas->generate_random_matrix_data(); - if (!copy_matrix) { - // copy matrix only once - if (!gpu_blas->copy_data_to_gpu()) { - *error = 1; - *err_description = PERF_BLAS_MEMCPY_ERROR; - } - } -} - - -/** - * @brief logs the Gflops computed over the last log_interval period - * @param gflops_interval the Gflops that the GPU achieved - */ -void PERFWorker::check_target_stress(double gflops_interval) { - string msg; - bool result; - rvs::action_result_t action_result; - - if(gflops_interval >= target_stress){ - result = true; - }else{ - result = false; - } - - msg = "[" + action_name + "] " + MODULE_NAME + " " + - std::to_string(gpu_id) + " " + PERF_LOG_GFLOPS_INTERVAL_KEY + " " + std::to_string(gflops_interval) + " " + - "Target stress :" + " " + std::to_string(target_stress) + " met :" + (result ? "TRUE" : "FALSE"); - rvs::lp::Log(msg, rvs::logresults); - - action_result.state = rvs::actionstate::ACTION_RUNNING; - action_result.status = (true == result) ? rvs::actionstatus::ACTION_SUCCESS : rvs::actionstatus::ACTION_FAILED; - action_result.output = msg.c_str(); - action.action_callback(&action_result); - - log_to_json(PERF_LOG_GFLOPS_INTERVAL_KEY, std::to_string(gflops_interval), - rvs::loginfo); -} - -/** - * @brief logs the Gflops computed over the last log_interval period - * @param gflops_interval the Gflops that the GPU achieved - */ -void PERFWorker::log_interval_gflops(double gflops_interval) { - string msg; - rvs::action_result_t action_result; - - msg = "[" + action_name + "] " + MODULE_NAME + " " + - std::to_string(gpu_id) + " " + PERF_LOG_GFLOPS_INTERVAL_KEY + " " + - std::to_string(gflops_interval); - rvs::lp::Log(msg, rvs::logresults); - - action_result.state = rvs::actionstate::ACTION_RUNNING; - action_result.status = rvs::actionstatus::ACTION_SUCCESS; - action_result.output = msg.c_str(); - action.action_callback(&action_result); - - log_to_json(PERF_LOG_GFLOPS_INTERVAL_KEY, std::to_string(gflops_interval), - rvs::loginfo); -} - - -/** - * @brief performs the stress test on the given GPU - * @param error pointer to a memory location where the error code will be stored - * @param err_description stores the error description if any - * @return true if stress violations is less than max_violations, false otherwise - */ -bool PERFWorker::do_perf_stress_test(int *error, std::string *err_description) { - double start_time, end_time; - double timetaken; - string msg; - - *error = 0; - max_gflops = 0; - start_time = 0; - end_time = 0; - - //Start the timer - start_time = gpu_blas->get_time_us(); - - // run GEMM & wait for completion - gpu_blas->run_blas_gemm(true); - - //End the timer - end_time = gpu_blas->get_time_us(); - - //Converting microseconds to seconds - timetaken = (end_time - start_time)/1e6; - - max_gflops = static_cast ((gpu_blas->gemm_gflop_count() * perf_hot_calls)/timetaken) ; - - log_interval_gflops(max_gflops); - - return true; -} - -/** - * @brief performs the stress test on the given GPU - */ -void PERFWorker::run() { - string msg, err_description; - int error = 0; - bool perf_test_passed = true; - - max_gflops = 0; - - // log PERF stress test - start message - msg = "[" + action_name + "] " + MODULE_NAME + " " + - std::to_string(gpu_id) + " " + PERF_START_MSG + " " + - " Starting the PERF stress test "; - rvs::lp::Log(msg, rvs::logtrace); - - log_to_json(PERF_START_MSG, std::to_string(target_stress), rvs::loginfo); - log_to_json(PERF_COPY_MATRIX_MSG, (copy_matrix ? "true":"false"), - rvs::loginfo); - - // stage 1. setup rvs blas - setup_blas(&error, &err_description); - if (error) - return; - - if (run_duration_ms > 0) { - perf_test_passed = do_perf_stress_test(&error, &err_description); - // check if stop signal was received - if (rvs::lp::Stopping()) - return; - - if (error) { - // GPU didn't complete the test (HIP/rocBlas error(s) occurred) - string msg = "[" + action_name + "] " + MODULE_NAME + " " + - std::to_string(gpu_id) + " " + err_description; - rvs::lp::Log(msg, rvs::logerror); - log_to_json("err", err_description, rvs::logerror); - return; - } - } - - log_interval_gflops(max_gflops); - check_target_stress(max_gflops); -} - - -/** - * @brief computes the difference (in milliseconds) between 2 points in time - * @param t_end second point in time - * @param t_start first point in time - * @return time difference in milliseconds - */ -uint64_t PERFWorker::time_diff( - std::chrono::time_point t_end, - std::chrono::time_point t_start) { - auto milliseconds = std::chrono::duration_cast( - t_end - t_start); - return milliseconds.count(); -} - -/** - * @brief logs a message to JSON - * @param key info type - * @param value message to log - * @param log_level the level of log (e.g.: info, results, error) - */ -void PERFWorker::log_to_json(const std::string &key, const std::string &value, - int log_level) { - if (PERFWorker::bjson) { - unsigned int sec; - unsigned int usec; - - rvs::lp::get_ticks(&sec, &usec); - void *json_node = rvs::lp::LogRecordCreate(MODULE_NAME, - action_name.c_str(), log_level, sec, usec); - if (json_node) { - rvs::lp::AddString(json_node, PERF_JSON_LOG_GPU_ID_KEY, std::to_string(gpu_id)); - - uint16_t gpu_index = 0; - rvs::gpulist::gpu2gpuindex(gpu_id, &gpu_index); - rvs::lp::AddString(json_node, "gpu_index", std::to_string(gpu_index)); - - rvs::lp::AddString(json_node, key, value); - rvs::lp::LogRecordFlush(json_node); - } - } -} - -/** - * @brief extends the usleep for more than 1000000us - * @param microseconds us to sleep - */ -void PERFWorker::usleep_ex(uint64_t microseconds) { - uint64_t total_microseconds = microseconds; - for (;;) { - if (total_microseconds > USLEEP_MAX_VAL) { - usleep(USLEEP_MAX_VAL); - total_microseconds -= USLEEP_MAX_VAL; - } else { - usleep(total_microseconds); - return; - } - } -} diff --git a/pesm.so/.gitignore b/pesm.so/.gitignore deleted file mode 100644 index bfe6c3ed1..000000000 --- a/pesm.so/.gitignore +++ /dev/null @@ -1,9 +0,0 @@ -/.settings/ -/CMakeFiles/ -/Debug/ -/build/ -/cmake_install.cmake -/Makefile -/.project -/lib*.so.*.*.* - diff --git a/pesm.so/CMakeLists.txt b/pesm.so/CMakeLists.txt deleted file mode 100644 index a04f2fd0c..000000000 --- a/pesm.so/CMakeLists.txt +++ /dev/null @@ -1,144 +0,0 @@ -################################################################################ -## -## Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. -## -## MIT LICENSE: -## Permission is hereby granted, free of charge, to any person obtaining a copy of -## this software and associated documentation files (the "Software"), to deal in -## the Software without restriction, including without limitation the rights to -## use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies -## of the Software, and to permit persons to whom the Software is furnished to do -## so, subject to the following conditions: -## -## The above copyright notice and this permission notice shall be included in all -## copies or substantial portions of the Software. -## -## THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -## IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -## FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -## AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -## LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -## OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE -## SOFTWARE. -## -################################################################################ - -cmake_minimum_required ( VERSION 3.5.0 ) -if ( ${CMAKE_BINARY_DIR} STREQUAL ${CMAKE_CURRENT_SOURCE_DIR}) - message(FATAL "In-source build is not allowed") -endif () -set (CMAKE_RUNTIME_OUTPUT_DIRECTORY "${CMAKE_BINARY_DIR}/bin") - -set ( RVS "pesm" ) -set ( RVS_PACKAGE "rvs-roct" ) -set ( RVS_COMPONENT "lib${RVS}" ) -set ( RVS_TARGET "${RVS}" ) - -project ( ${RVS_TARGET} ) - -message(STATUS "MODULE: ${RVS}") - -## Set default module path if not already set -add_compile_options(-pthread) -add_compile_options(-Wall) -if (RVS_COVERAGE) - add_compile_options(-o0 -fprofile-arcs -ftest-coverage) - set(CMAKE_EXE_LINKER_FLAGS "--coverage") - set(CMAKE_SHARED_LINKER_FLAGS "--coverage") -endif() - -if ( NOT DEFINED CMAKE_MODULE_PATH ) - set ( CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/../cmake_modules/" ) -endif () - -## Include common cmake modules -include ( utils ) - -## Setup the package version. -get_version ( "0.0.0" ) - -set ( BUILD_VERSION_MAJOR ${VERSION_MAJOR} ) -set ( BUILD_VERSION_MINOR ${VERSION_MINOR} ) -set ( BUILD_VERSION_PATCH ${VERSION_PATCH} ) -set ( LIB_VERSION_STRING "${BUILD_VERSION_MAJOR}.${BUILD_VERSION_MINOR}.${BUILD_VERSION_PATCH}" ) - -if ( DEFINED VERSION_BUILD AND NOT ${VERSION_BUILD} STREQUAL "" ) - set ( BUILD_VERSION_PATCH "${BUILD_VERSION_PATCH}-${VERSION_BUILD}" ) -endif () -set ( BUILD_VERSION_STRING "${BUILD_VERSION_MAJOR}.${BUILD_VERSION_MINOR}.${BUILD_VERSION_PATCH}" ) - -## make version numbers visible to C code -add_compile_options(-DBUILD_VERSION_MAJOR=${VERSION_MAJOR}) -add_compile_options(-DBUILD_VERSION_MINOR=${VERSION_MINOR}) -add_compile_options(-DBUILD_VERSION_PATCH=${VERSION_PATCH}) -add_compile_options(-DLIB_VERSION_STRING="${LIB_VERSION_STRING}") -add_compile_options(-DBUILD_VERSION_STRING="${BUILD_VERSION_STRING}") - -# Determine HSA_PATH -if(NOT DEFINED HIPCC_PATH) - if(NOT DEFINED ENV{HIPCC_PATH}) - set(HIPCC_PATH "${ROCM_PATH}" CACHE PATH "Path to which hipcc runtime has been installed") - else() - set(HIPCC_PATH $ENV{HIPCC_PATH} CACHE PATH "Path to which hipcc runtime has been installed") - endif() -endif() - -# Add HIP_VERSION to CMAKE__FLAGS -set(HIP_HCC_BUILD_FLAGS "${HIP_HCC_BUILD_FLAGS} -DHIP_VERSION_MAJOR=${HIP_VERSION_MAJOR} -DHIP_VERSION_MINOR=${HIP_VERSION_MINOR} -DHIP_VERSION_PATCH=${HIP_VERSION_GITDATE}") - -set(HIP_HCC_BUILD_FLAGS) -set(HIP_HCC_BUILD_FLAGS "${HIP_HCC_BUILD_FLAGS} -fPIC ${HCC_CXX_FLAGS} -I${HSA_INC_DIR} ${ASAN_CXX_FLAGS}") - -# Set compiler and compiler flags -set(CMAKE_CXX_COMPILER "${HIPCC_PATH}/bin/hipcc") -set(CMAKE_C_COMPILER "${HIPCC_PATH}/bin/hipcc") -set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} ${HIP_HCC_BUILD_FLAGS}") -set(CMAKE_C_FLAGS "${CMAKE_C_FLAGS} ${HIP_HCC_BUILD_FLAGS}") -set(CMAKE_EXE_LINKER_FLAGS "${CMAKE_EXE_LINKER_FLAGS} ${ASAN_LD_FLAGS}") -set(CMAKE_SHARED_LINKER_FLAGS "${CMAKE_SHARED_LINKER_FLAGS} ${ASAN_LD_FLAGS}") - -if(BUILD_ADDRESS_SANITIZER) - execute_process(COMMAND ${CMAKE_CXX_COMPILER} --print-file-name=libclang_rt.asan-x86_64.so - OUTPUT_VARIABLE ASAN_LIB_FULL_PATH) - get_filename_component(ASAN_LIB_PATH ${ASAN_LIB_FULL_PATH} DIRECTORY) -else() - set(ASAN_LIB_PATH "$ENV{LD_LIBRARY_PATH}") -endif() - -## define include directories -include_directories(./ ../ pci) -# Add directories to look for library files to link -link_directories(${RVS_LIB_DIR} ${ROCR_LIB_DIR} ${ROCBLAS_LIB_DIR} ${ASAN_LIB_PATH} ${AMD_SMI_LIB_DIR} ${HIPRAND_DIR} ${ROCRAND_DIR}) -## additional libraries -set (PROJECT_LINK_LIBS libpthread.so libpci.so libm.so) - -## define source files -set(SOURCES src/rvs_module.cpp src/action.cpp src/worker.cpp) - -## define target -add_library( ${RVS_TARGET} SHARED ${SOURCES}) -set_target_properties(${RVS_TARGET} PROPERTIES - SUFFIX .so.${LIB_VERSION_STRING} - LIBRARY_OUTPUT_DIRECTORY ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}) -target_link_libraries(${RVS_TARGET} rvslib ${PROJECT_LINK_LIBS} ) -add_dependencies(${RVS_TARGET} rvslib) - -add_custom_command(TARGET ${RVS_TARGET} POST_BUILD -COMMAND ln -fs ./lib${RVS}.so.${LIB_VERSION_STRING} lib${RVS}.so.${VERSION_MAJOR} WORKING_DIRECTORY ${CMAKE_RUNTIME_OUTPUT_DIRECTORY} -COMMAND ln -fs ./lib${RVS}.so.${VERSION_MAJOR} lib${RVS}.so WORKING_DIRECTORY ${CMAKE_RUNTIME_OUTPUT_DIRECTORY} -) - -install(TARGETS ${RVS_TARGET} LIBRARY DESTINATION ${CPACK_PACKAGING_INSTALL_PREFIX}/${CMAKE_INSTALL_LIBDIR}/rvs COMPONENT rvsmodule) -install(FILES "${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/lib${RVS}.so.${VERSION_MAJOR}" - DESTINATION ${CPACK_PACKAGING_INSTALL_PREFIX}/${CMAKE_INSTALL_LIBDIR}/rvs COMPONENT rvsmodule) -install(FILES "${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/lib${RVS}.so" - DESTINATION ${CPACK_PACKAGING_INSTALL_PREFIX}/${CMAKE_INSTALL_LIBDIR}/rvs COMPONENT rvsmodule) - -# TEST SECTION -if (RVS_BUILD_TESTS) - add_custom_command(TARGET ${RVS_TARGET} POST_BUILD - COMMAND ln -fs ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/lib${RVS}.so.${VERSION_MAJOR} ${RVS_BINTEST_FOLDER}/lib${RVS}.so WORKING_DIRECTORY ${CMAKE_RUNTIME_OUTPUT_DIRECTORY} - ) - include(${CMAKE_CURRENT_SOURCE_DIR}/tests.cmake) -endif() - diff --git a/pesm.so/include/.gitignore b/pesm.so/include/.gitignore deleted file mode 100644 index e69de29bb..000000000 diff --git a/pesm.so/include/action.h b/pesm.so/include/action.h deleted file mode 100644 index b7ac7be5d..000000000 --- a/pesm.so/include/action.h +++ /dev/null @@ -1,63 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#ifndef PESM_SO_INCLUDE_ACTION_H_ -#define PESM_SO_INCLUDE_ACTION_H_ - -#include -#include - -#include "include/rvsactionbase.h" - -/** - * @class pesm_action - * @ingroup PESM - * - * @brief PESM action implementation class - * - * Derives from rvs::actionbase and implements actual action functionality - * in its run() method. - * - */ -class pesm_action : public rvs::actionbase { - public: - pesm_action(); - virtual ~pesm_action(); - - virtual int run(void); - protected: - int do_gpu_list(void); - bool get_all_common_config_keys() override; - bool get_all_pesm_config_keys(void); - - protected: - - friend class Worker; - //! debug wait helper - int prop_debugwait; - //! 'true' if monitoring is to be initiated - bool prop_monitor; -}; - -#endif // PESM_SO_INCLUDE_ACTION_H_ diff --git a/pesm.so/include/rvs_module.h b/pesm.so/include/rvs_module.h deleted file mode 100644 index 7896a64ac..000000000 --- a/pesm.so/include/rvs_module.h +++ /dev/null @@ -1,30 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#ifndef PESM_SO_INCLUDE_RVS_MODULE_H_ -#define PESM_SO_INCLUDE_RVS_MODULE_H_ - -#include "include/rvsliblog.h" - -#endif // PESM_SO_INCLUDE_RVS_MODULE_H_ diff --git a/pesm.so/include/worker.h b/pesm.so/include/worker.h deleted file mode 100644 index 17981c469..000000000 --- a/pesm.so/include/worker.h +++ /dev/null @@ -1,102 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#ifndef PESM_SO_INCLUDE_WORKER_H_ -#define PESM_SO_INCLUDE_WORKER_H_ - -#include -#include - -#include "include/rvsthreadbase.h" -#include "include/rvsactionbase.h" -#include "include/action.h" - - -/** - * @class Worker - * @ingroup PESM - * - * @brief Monitoring implementation class - * - * Derives from rvs::ThreadBase and implements actual monitoring functionality - * in its run() method. - * - */ - -class Worker : public rvs::ThreadBase { - public: - Worker(); - virtual ~Worker(); - - //! Stops monitoring - void stop(void); - //! Sets initiating action name - void set_name(const std::string& name) { action_name = name; } - //! sets action - void set_action(const pesm_action& _action) { action = _action; } - //! sets stopping action name - void set_stop_name(const std::string& name) { stop_action_name = name; } - //! Sets device id for filtering - void set_deviceid(const int id) { device_id = id; } - //! Sets GPU IDs for filtering - void set_gpuids(const std::vector& GpuIds); - //! Sets GPU indexes for filtering - void set_gpuidx(const std::vector& GpuIdx); - //! Sets GPU IDs for filtering (string used in messages) - //! @param Devices List of devices to monitor - void set_strgpuids(const std::string& Devices) { strgpuids = Devices; } - //! Sets JSON flag - void json(const bool flag) { bjson = flag; } - //! Returns initiating action name - const std::string& get_name(void) { return action_name; } - - protected: - virtual void run(void); - - protected: - //! TRUE if JSON output is required - bool bjson; - //! Loops while TRUE - bool brun; - //! device id to filter for. 0 if no filtering. - int device_id; - //! GPU id filtering flag - bool bfiltergpu; - //! GPU indexes filtering flag - bool bfiltergpuidx; - //! list of GPU devices to monitor - std::vector gpuids; - //! list of GPU device indexes to monitor - std::vector gpuidx; - //! list of GPU devices to monitor (string used in messages) - std::string strgpuids; - //! Name of the action which initiated monitoring - std::string action_name; - //! action instance - pesm_action action; - //! Name of the action which stops monitoring - std::string stop_action_name; -}; - -#endif // PESM_SO_INCLUDE_WORKER_H_ diff --git a/pesm.so/src/.gitignore b/pesm.so/src/.gitignore deleted file mode 100644 index 6677c8735..000000000 --- a/pesm.so/src/.gitignore +++ /dev/null @@ -1 +0,0 @@ -/libmain.cpp diff --git a/pesm.so/src/action.cpp b/pesm.so/src/action.cpp deleted file mode 100644 index ec6cb336d..000000000 --- a/pesm.so/src/action.cpp +++ /dev/null @@ -1,265 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#include "include/action.h" - -extern "C" { -#include -#include -} - -#include -#include -#include -#include -#include -#include - -#include "include/rvs_key_def.h" -#include "include/rvs_module.h" -#include "include/worker.h" -#include "include/pci_caps.h" -#include "include/gpu_util.h" -#include "include/rvs_util.h" -#include "include/rvsloglp.h" -#define RVS_CONF_DBGWAIT_KEY "debugwait" - -static constexpr auto MODULE_NAME = "pesm"; -static constexpr auto MODULE_NAME_CAPS = "PESM"; -using std::string; -using std::cout; -using std::endl; -using std::hex; - - -extern Worker* pworker; - -//! Default constructor -pesm_action::pesm_action() { - bjson = false; - prop_monitor = true; - module_name = MODULE_NAME; -} - -//! Default destructor -pesm_action::~pesm_action() { - property.clear(); -} - -/** - * @brief reads all common configuration keys from - * the module's properties collection - * @return true if no fatal error occured, false otherwise - */ -bool pesm_action::get_all_common_config_keys(void) { - string msg; - - bool sts = true; - - if (property_get(RVS_CONF_NAME_KEY, &action_name)) { - rvs::lp::Err("Action name missing", MODULE_NAME_CAPS); - return false; - } - - // check if -j flag is passed - if (has_property("cli.-j")) { - bjson = true; - } - - // get property value (a list of gpu id) - if (int ists = property_get_device()) { - switch (ists) { - case 1: - msg = "Invalid 'device' key value."; - break; - case 2: - msg = "Missing 'device' key."; - break; - } - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - sts = false; - } - - // get the property value if provided - if (property_get_int(RVS_CONF_DEVICEID_KEY, - &property_device_id, 0u)) { - msg = "Invalid 'deviceid' key value."; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - sts = false; - } - - // get property value (a list of device indexes) - if (int sts = property_get_device_index()) { - switch (sts) { - case 1: - msg = "Invalid 'device_index' key value."; - break; - case 2: - msg = "Missing 'device_index' key."; - break; - } - // default set as true - property_device_index_all = true; - rvs::lp::Log(msg, rvs::loginfo); - } - - return sts; -} - -/** - * @brief reads all PESM specific configuration keys from - * the module's properties collection - * @return true if no fatal error occured, false otherwise - */ -bool pesm_action::get_all_pesm_config_keys(void) { - string msg; - - bool sts = true; - - // get the property value if provided - if (property_get(RVS_CONF_MONITOR_KEY, &prop_monitor, true)) { - msg = "Invalid '" RVS_CONF_MONITOR_KEY "' key value."; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - sts = false; - } - - // get the property value if provided - if (property_get_int(RVS_CONF_DBGWAIT_KEY, &prop_debugwait, 0)) { - msg = "Invalid '" RVS_CONF_DBGWAIT_KEY "' key value."; - rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); - sts = false; - } - - return sts; -} - - -/** - * @brief Implements action functionality - * - * Functionality: - * - * - If "do_gpu_list" property is set, - * it lists all AMD GPUs present in the system and exits - * - If "monitor" property is set to "true", - * it creates Worker thread and initiates monitoring and exits - * - If "monitor" property is not set or is not set to "true", - * it stops the Worker thread and exits - * - * @return 0 - success. non-zero otherwise - * - * */ -int pesm_action::run(void) { - string msg; - RVSTRACE_ - - // this module implements --listGpu command line option - // if this option is set, an internal input key 'do_gpu_list' is passed - // to this action - if (has_property("do_gpu_list")) { - return do_gpu_list(); - } - - // get commong configuration keys - if (!get_all_common_config_keys()) { - return 1; - } - - // get PESM specific configuration keys - if (!get_all_pesm_config_keys()) { - return 1; - } - - // debugging help - if (prop_debugwait) { - sleep(prop_debugwait); - } - if (bjson){ - json_add_primary_fields(std::string(MODULE_NAME), action_name); - } - // end of monitoring requested? - if (!prop_monitor) { - RVSTRACE_ - if (pworker) { - RVSTRACE_ - // (give thread chance to start) - sleep(2); - pworker->set_stop_name(action_name); - pworker->stop(); - delete pworker; - pworker = nullptr; - } - if (bjson){ - rvs::lp::JsonActionEndNodeCreate(); - } - RVSTRACE_ - return 0; - } - - RVSTRACE_ - if (pworker) { - rvs::lp::Log("[" + property["name"]+ "] pesm monitoring already started", - rvs::logdebug); - if (bjson){ - rvs::lp::JsonActionEndNodeCreate(); - } - - return 0; - } - - RVSTRACE_ - // create worker thread - pworker = new Worker(); - pworker->set_name(action_name); - pworker->set_action(*this); - pworker->json(bjson); - pworker->set_gpuids(property_device); - pworker->set_gpuidx(property_device_index); - pworker->set_deviceid(property_device_id); - - // start worker thread - RVSTRACE_ - pworker->start(); - sleep(2); - if (bjson){ - rvs::lp::JsonActionEndNodeCreate(); - } - RVSTRACE_ - return 0; -} - -/** - * @brief Lists AMD GPUs - * - * Functionality: - * - * Lists all AMD GPUs present in the system. - * - * @return 0 - success. non-zero otherwise - * - * */ -int pesm_action::do_gpu_list() { - return display_gpu_info(); -} - diff --git a/pesm.so/src/rvs_module.cpp b/pesm.so/src/rvs_module.cpp deleted file mode 100644 index 4f468c2df..000000000 --- a/pesm.so/src/rvs_module.cpp +++ /dev/null @@ -1,123 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#include "include/rvs_module.h" - -#include -#include -#include - -#include "include/gpu_util.h" -#include "include/rvsloglp.h" -#include "include/worker.h" -#include "include/action.h" - -/** - * @defgroup PESM PESM Module - * - * @brief PCIe State Monitoring module - * - * The PCIe State Monitor tool is used to actively monitor the PCIe interconnect between the host - * platform and the GPU. The module will register a “listener” on a target GPU’s PCIe - * interconnect, and log a message whenever it detects a state change. The PESM will be able to - * detect the following state changes: - * - 1.2.PCIe link speed changes - * - GPU power state changes - */ - -Worker* pworker; - -extern "C" int rvs_module_has_interface(int iid) { - int sts = 0; - switch (iid) { - case 0: - case 1: - sts = 1; - } - return sts; -} - -extern "C" const char* rvs_module_get_description(void) { - return "The PCIe State Monitor tool is used to actively monitor the PCIe interconnect between the host platform and the GPU."; -} - -extern "C" const char* rvs_module_get_config(void) { - return "monitor (bool)"; -} - -extern "C" const char* rvs_module_get_output(void) { - return "state (string)"; -} - -extern "C" int rvs_module_init(void* pMi) { - pworker = nullptr; - rvs::lp::Initialize(static_cast(pMi)); - rvs::gpulist::Initialize(); - return 0; -} - -extern "C" int rvs_module_terminate(void) { - rvs::lp::Log("[module_terminate] pesm rvs_module_terminate() - entered", - rvs::logtrace); - if (pworker) { - rvs::lp::Log( - "[module_terminate] pesm rvs_module_terminate() - pworker exists", - rvs::logtrace); - pworker->set_stop_name("module_terminate"); - pworker->stop(); - delete pworker; - pworker = nullptr; - rvs::lp::Log( - "[module_terminate] pesm rvs_module_terminate() - monitoring stopped", - rvs::logtrace); - } - cleanup_logs(); - amdsmi_shut_down(); - return 0; -} - -extern "C" void* rvs_module_action_create(void) { - return static_cast(new pesm_action); -} - -extern "C" int rvs_module_action_destroy(void* pAction) { - delete static_cast(pAction); - return 0; -} - -extern "C" int rvs_module_action_property_set( - void* pAction, const char* Key, const char* Val) { - return static_cast(pAction)->property_set(Key, Val); -} - -extern "C" int rvs_module_action_callback_set(void* pAction, - rvs::callback_t callback, - void * user_param) { - return static_cast(pAction)->callback_set(callback, user_param); -} - -extern "C" int rvs_module_action_run(void* pAction) { - return static_cast(pAction)->run(); -} - diff --git a/pesm.so/src/worker.cpp b/pesm.so/src/worker.cpp deleted file mode 100644 index 8915903f8..000000000 --- a/pesm.so/src/worker.cpp +++ /dev/null @@ -1,335 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#include "include/worker.h" - -#include -#include -#include -#include -#include -#include - -#ifdef __cplusplus -extern "C" { -#endif -#include -#include -#ifdef __cplusplus -} -#endif - -#include "include/rvs_module.h" -#include "include/pci_caps.h" -#include "include/gpu_util.h" -#include "include/rvs_util.h" -#include "include/rvsloglp.h" -#define MODULE_NAME "PESM" - -using std::string; -using std::vector; -using std::map; - -Worker::Worker() { - bfiltergpu = false; - bfiltergpuidx = false; -} -Worker::~Worker() {} - -/** - * @brief Sets GPU IDs for filtering - * @arg GpuIds Array of GPU GpuIds - */ -void Worker::set_gpuids(const std::vector& GpuIds) { - gpuids = GpuIds; - if (gpuids.size()) { - bfiltergpu = true; - } -} - -/** - * @brief Sets GPU Indexes for filtering - * @arg GpuIdx Array of GPU indexes - */ -void Worker::set_gpuidx(const std::vector& GpuIdx) { - gpuidx = GpuIdx; - if (gpuidx.size()) { - bfiltergpuidx = true; - } -} - -/** - * @brief Thread function - * - * Loops while brun == TRUE and performs polled monitoring every 1msec. - * - * */ -void Worker::run() { - char buff[1024]; - - map::iterator it; - vector gpus_location_id; - map old_speed_val; - map old_pwr_val; - - struct pci_access *pacc; - struct pci_dev *dev; - - unsigned int sec; - unsigned int usec; - void* r; - rvs::action_result_t action_result; - map speed_change; - map power_change; - string msg; - - brun = true; - - // get timestamp - rvs::lp::get_ticks(&sec, &usec); - - // add string output - msg = "[" + action_name + "] " + "PCIe link speed and power monitoring started ..."; - rvs::lp::Log(msg, rvs::logresults, sec, usec); - - if (bjson) { - // add JSON output - r = json_node_create("pesm", action_name.c_str(), rvs::logresults); - rvs::lp::AddString(r, "msg", "started"); - rvs::lp::LogRecordFlush(r); - } - - // worker thread has started - while (brun) { - rvs::lp::Log("[" + action_name + "] pesm worker thread is running...", - rvs::logtrace); - - // get the pci_access structure - pacc = pci_alloc(); - // initialize the PCI library - pci_init(pacc); - // get the list of devices - pci_scan_bus(pacc); - - // iterate over devices - for (dev = pacc->devices; dev; dev = dev->next) { - - int known_fields = pci_fill_info(dev, PCI_FILL_IDENT | PCI_FILL_BASES | PCI_FILL_CLASS - | PCI_FILL_EXT_CAPS | PCI_FILL_CAPS | PCI_FILL_PHYS_SLOT); // fil in the info - - // computes the actual dev's location_id (sysfs entry) - uint16_t dev_location_id = - ((((uint16_t)(dev->bus)) << 8) | (((uint16_t)(dev->dev)) << 3) | ((uint16_t)(dev->func))) ; - - uint16_t gpu_id; - // if not and AMD GPU just continue - if (rvs::gpulist::location2gpu(dev_location_id, &gpu_id)) - continue; - - uint16_t gpu_idx; - if (rvs::gpulist::gpu2gpuindex(gpu_id, &gpu_idx)) - continue; - - // device_id filtering - if ( device_id != 0 && dev->device_id != device_id) - continue; - - // gpu id filtering - if (bfiltergpu) { - auto itgpuid = find(gpuids.begin(), gpuids.end(), gpu_id); - if (itgpuid == gpuids.end()) - continue; - } - - // gpu index filtering - if (bfiltergpuidx) { - auto itgpuidx = find(gpuidx.begin(), gpuidx.end(), gpu_idx); - if (itgpuidx == gpuidx.end()) - continue; - } - - rvs::lp::get_ticks(&sec, &usec); - - // get current speed for the link - get_link_stat_cur_speed(dev, buff); - string new_speed_val(buff); - if(old_speed_val[gpu_id].empty()) { - old_speed_val[gpu_id] = new_speed_val; - } - - // get current power state for GPU - get_pwr_curr_state(dev, buff); - string new_pwr_val(buff); - if(old_pwr_val[gpu_id].empty()) { - old_pwr_val[gpu_id] = new_pwr_val; - continue; - } - - // link speed changed - if (old_speed_val[gpu_id] != new_speed_val) { - // new value is different, so store it; - old_speed_val[gpu_id] = new_speed_val; - - msg = "[" + action_name + "] " + std::to_string(gpu_id) + - " PCIe link speed changed " + new_speed_val; - rvs::lp::Log(msg, rvs::loginfo, sec, usec); - - action_result.state = rvs::actionstate::ACTION_RUNNING; - action_result.status = rvs::actionstatus::ACTION_SUCCESS; - action_result.output = msg.c_str(); - action.action_callback(&action_result); - - speed_change[gpu_id] = "true"; - } - else { - msg = "[" + action_name + "] " + "pesm " + - std::to_string(gpu_id) + " PCIe link speed unchanged " + new_speed_val; - rvs::lp::Log(msg, rvs::loginfo, sec, usec); - - if(speed_change[gpu_id].empty()) { - speed_change[gpu_id] = "false"; - } - } - - // power state changed - if (old_pwr_val[gpu_id] != new_pwr_val) { - // new value is different, so store it; - old_pwr_val[gpu_id] = new_pwr_val; - - msg = "[" + action_name + "] " + std::to_string(gpu_id) - + " PCIe power state changed " + new_pwr_val; - rvs::lp::Log(msg, rvs::loginfo, sec, usec); - - action_result.state = rvs::actionstate::ACTION_RUNNING; - action_result.status = rvs::actionstatus::ACTION_SUCCESS; - action_result.output = msg.c_str(); - action.action_callback(&action_result); - - power_change[gpu_id] = "true"; - } - else { - msg = "[" + action_name + "] " + - std::to_string(gpu_id) + " PCIe power state unchanged " + new_pwr_val; - rvs::lp::Log(msg, rvs::loginfo, sec, usec); - - if(power_change[gpu_id].empty()) { - power_change[gpu_id] = "false"; - } - } - } - - pci_cleanup(pacc); - - sleep(1); - } - - // get timestamp - rvs::lp::get_ticks(&sec, &usec); - - string gpu_json; - string speed_json; - string power_json; - - for(auto i : speed_change) { - - msg = "[" + stop_action_name + "]" + - " GPU " + std::to_string(i.first) + " PCIe speed change " + i.second; - - rvs::lp::Log(msg, rvs::logresults, sec, usec); - - if (bjson) { - r = json_node_create("PESM", - stop_action_name.c_str(), rvs::logresults); - rvs::lp::AddString(r, "gpu_id", std::to_string(i.first)); - - uint16_t gpu_index = 0; - rvs::gpulist::gpu2gpuindex(i.first, &gpu_index); - rvs::lp::AddString(r, "gpu_index", std::to_string(gpu_index)); - - rvs::lp::AddString(r, "pcie speed change",i.second); - rvs::lp::LogRecordFlush(r); - r = nullptr; - } - } - - - for(auto i : power_change) { - - msg = "[" + stop_action_name + "]" + - " GPU " + std::to_string(i.first) + " PCIe power change " + i.second; - rvs::lp::Log(msg, rvs::logresults, sec, usec); - if (bjson){ - r = json_node_create("PESM", - stop_action_name.c_str(), rvs::logresults); - rvs::lp::AddString(r, "gpu_id", std::to_string(i.first)); - - uint16_t gpu_index = 0; - rvs::gpulist::gpu2gpuindex(i.first, &gpu_index); - rvs::lp::AddString(r, "gpu_index", std::to_string(gpu_index)); - - rvs::lp::AddString(r, "power_change", i.second); - rvs::lp::AddString(r, "msg", "stopped"); - rvs::lp::LogRecordFlush(r); - r = nullptr; - } - } - - // add string output - msg = "[" + stop_action_name + "] PCIe monitoring ended after wait duration."; - rvs::lp::Log(msg, rvs::logresults, sec, usec); - - action_result.state = rvs::actionstate::ACTION_COMPLETED; - action_result.status = rvs::actionstatus::ACTION_SUCCESS; - action_result.output = msg.c_str(); - action.action_callback(&action_result); - - - rvs::lp::Log("[" + stop_action_name + "] pesm worker thread has finished", - rvs::logdebug); -} - -/** - * @brief Stops monitoring - * - * Sets brun member to FALSE thus signaling end of monitoring. - * Then it waits for std::thread to exit before returning. - * - * */ -void Worker::stop() { - rvs::lp::Log("[" + stop_action_name + "] pesm in Worker::stop()", - rvs::logtrace); - // reset "run" flag - brun = false; - // (give thread chance to finish processing and exit) - sleep(200); - - // wait a bit to make sure thread has exited - try { - if (t.joinable()) - t.join(); - } - catch(...) { - } -} - diff --git a/pesm.so/test/test_actionbase.cpp b/pesm.so/test/test_actionbase.cpp deleted file mode 100644 index a599f8081..000000000 --- a/pesm.so/test/test_actionbase.cpp +++ /dev/null @@ -1,159 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ - -#include -#include - -#include "gtest/gtest.h" - -#include "test/unitactionbase.h" - -TEST(actionbase, run) { - rvs::actionbase* p = new unitactionbase; - - EXPECT_EQ(p->run(), 3); - delete p; -} - -TEST(actionbase, boolttp) { - std::string val; - bool bretval; - int intretval; - bool bval; - - rvs::actionbase* p = new unitactionbase; - assert(p); - - p->property_set("bool1", "true"); - p->property_set("bool2", "false"); - bretval = p->has_property("bool1", &val); - EXPECT_EQ(bretval, true); - EXPECT_EQ(val, "true"); - - intretval = p->property_get("bool1", &bval); - EXPECT_EQ(bval, true); - EXPECT_EQ(intretval, 0); - - intretval = p->property_get("bool2", &bval); - EXPECT_EQ(bval, false); - EXPECT_EQ(intretval, 0); - - delete p; -} - -TEST(actionbase, boolttf) { - int intretval; - bool bval; - - rvs::actionbase* p = new unitactionbase; - assert(p); - - p->property_set("bool3", "xxx"); - p->property_set("bool4", "TRUE"); - p->property_set("bool5", "FALSE"); - - intretval = p->property_get("bool3", &bval); - EXPECT_EQ(intretval, 1); - - intretval = p->property_get("bool4", &bval); - EXPECT_EQ(intretval, 1); - - intretval = p->property_get("bool5", &bval); - EXPECT_EQ(intretval, 1); - - intretval = p->property_get("bool6", &bval); - EXPECT_EQ(intretval, 2); - - delete p; -} - -TEST(actionbase, uintttf) { - int intretval; - uint64_t intval; - - rvs::actionbase* p = new unitactionbase; - assert(p); - - p->property_set("uint1", "123456"); - p->property_set("uint2", "abcd"); - p->property_set("uint3", "-123"); - - intretval = p->property_get_int("uint1", &intval); - EXPECT_EQ(intretval, 0); - EXPECT_EQ(intval, 123456u); - - intretval = p->property_get_int("uint2", &intval); - EXPECT_EQ(intretval, 1); - - intretval = p->property_get_int("uint3", &intval); - EXPECT_EQ(intretval, 1); - - intretval = p->property_get_int("uint4", &intval); - EXPECT_EQ(intretval, 2); - - delete p; -} - -TEST(actionbase, devicetf) { - int intretval; - bool b_all; - std::vector dev; - - unitactionbase* p = new unitactionbase; - assert(p); - - p->property_set("device", "26720"); - - intretval = p->property_get_device(); - EXPECT_EQ(intretval, 0); - p->test_get_device_all(&dev, &b_all); - EXPECT_EQ(dev[0], 26720u); - - p->test_erase_property("device"); - p->property_set("device", "123 456"); - intretval = p->property_get_device(); - p->test_get_device_all(&dev, &b_all); - EXPECT_EQ(intretval, 0); - EXPECT_EQ(dev[0], 123u); - EXPECT_EQ(dev[1], 456u); - EXPECT_EQ(dev.size(), 2u); - - p->test_erase_property("device"); - p->property_set("device", "all"); - intretval = p->property_get_device(); - p->test_get_device_all(&dev, &b_all); - EXPECT_EQ(intretval, 0); - EXPECT_TRUE(b_all); - - p->test_erase_property("device"); - p->property_set("device", "abcd"); - intretval = p->property_get_device(); - p->test_get_device_all(&dev, &b_all); - EXPECT_EQ(intretval, 1); - EXPECT_FALSE(b_all); - EXPECT_EQ(dev.size(), 0u); - - delete p; -} diff --git a/pesm.so/test/test_sanity.cpp b/pesm.so/test/test_sanity.cpp deleted file mode 100644 index f18f7211d..000000000 --- a/pesm.so/test/test_sanity.cpp +++ /dev/null @@ -1,34 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ - - - -#include "gtest/gtest.h" - - -TEST(pesm, sanity) { - EXPECT_EQ(strlen("Test"), 4u); - EXPECT_EQ(strlen(""), 0u); -} diff --git a/pesm.so/test/unitactionbase.cpp b/pesm.so/test/unitactionbase.cpp deleted file mode 100644 index 0f8157c9a..000000000 --- a/pesm.so/test/unitactionbase.cpp +++ /dev/null @@ -1,57 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#include -#include - -#include "test/unitactionbase.h" -#include "include/rvsloglp.h" - -#define MODULE_NAME_CAPS "UNITPESM" - -//! Default constructor -unitactionbase::unitactionbase() { -} - -//! Default destructor -unitactionbase::~unitactionbase() { - property.clear(); -} - - -int unitactionbase::run(void) { - RVSTRACE_ - // agreed upon return value for this run() method - return 3; -} - -void unitactionbase::test_get_device_all -(std::vector* pDev, bool* bAll) { - *pDev = property_device; - *bAll = property_device_all; -} - -void unitactionbase::test_erase_property(const std::string& prop) { - property.erase(prop); -} diff --git a/pesm.so/test/unitactionbase.h b/pesm.so/test/unitactionbase.h deleted file mode 100644 index f395b96bf..000000000 --- a/pesm.so/test/unitactionbase.h +++ /dev/null @@ -1,44 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#ifndef PESM_SO_TEST_UNITACTIONBASE_H_ -#define PESM_SO_TEST_UNITACTIONBASE_H_ - -#include -#include - -#include "include/rvsactionbase.h" - -class unitactionbase : public rvs::actionbase { - public: - unitactionbase(); - virtual ~unitactionbase(); - - virtual int run(void); - - void test_get_device_all(std::vector* pDev, bool* bAll); - void test_erase_property(const std::string& prop); -}; - -#endif // PESM_SO_TEST_UNITACTIONBASE_H_ diff --git a/pesm.so/tests.cmake b/pesm.so/tests.cmake deleted file mode 100644 index 072fb98f5..000000000 --- a/pesm.so/tests.cmake +++ /dev/null @@ -1,49 +0,0 @@ -################################################################################ -## -## Copyright (c) 2018-2025 Advanced Micro Devices, Inc. All rights reserved. -## -## MIT LICENSE: -## Permission is hereby granted, free of charge, to any person obtaining a copy of -## this software and associated documentation files (the "Software"), to deal in -## the Software without restriction, including without limitation the rights to -## use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies -## of the Software, and to permit persons to whom the Software is furnished to do -## so, subject to the following conditions: -## -## The above copyright notice and this permission notice shall be included in all -## copies or substantial portions of the Software. -## -## THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -## IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -## FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -## AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -## LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -## OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE -## SOFTWARE. -## -################################################################################ - -set(ROCBLAS_LIB "rocblas") -set(HIPRAND_LIB "hiprand") -set(HIPBLASLT_LIB "hipblaslt") -set(CORE_RUNTIME_NAME "hsa-runtime") -set(CORE_RUNTIME_TARGET "${CORE_RUNTIME_NAME}64") - -find_package(OpenMP) - -set(UT_LINK_LIBS libpthread.so libpci.so libm.so libdl.so ${AMD_SMI_LIB} OpenMP::OpenMP_CXX - ${ROCBLAS_LIB} ${ROC_THUNK_NAME} ${CORE_RUNTIME_TARGET} ${ROCM_CORE} ${YAML_CPP_LIBRARIES} ${HIPRAND_LIB} ${HIPBLASLT_LIB} -) - -# Add directories to look for library files to link -link_directories(${AMD_SMI_LIB_DIR} ${ROCBLAS_LIB_DIR} ${HIPRAND_LIB_DIR} ${HIPBLASLT_LIB_DIR} ${YAML_CPP_LIBRARY_DIR}) - -set (UT_SOURCES test/unitactionbase.cpp -) - -# add unit tests -include(tests_unit) - -# Add configuration tests -include(tests_conf_logging) - diff --git a/perf.so/CMakeLists.txt b/pulse.so/CMakeLists.txt similarity index 89% rename from perf.so/CMakeLists.txt rename to pulse.so/CMakeLists.txt index 1f135ce23..01d2f4e53 100644 --- a/perf.so/CMakeLists.txt +++ b/pulse.so/CMakeLists.txt @@ -1,6 +1,6 @@ ################################################################################ ## -## Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. +## Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. ## ## MIT LICENSE: ## Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -29,7 +29,7 @@ if ( ${CMAKE_BINARY_DIR} STREQUAL ${CMAKE_CURRENT_SOURCE_DIR}) endif () set (CMAKE_RUNTIME_OUTPUT_DIRECTORY "${CMAKE_BINARY_DIR}/bin") -set ( RVS "perf" ) +set ( RVS "pulse" ) set ( RVS_PACKAGE "rvs-roct" ) set ( RVS_COMPONENT "lib${RVS}" ) set ( RVS_TARGET "${RVS}" ) @@ -37,7 +37,8 @@ set ( RVS_TARGET "${RVS}" ) project ( ${RVS_TARGET} ) message(STATUS "MODULE: ${RVS}") -add_compile_options(-Wall ) + +add_compile_options(-Wall) if (RVS_COVERAGE) add_compile_options(-o0 -fprofile-arcs -ftest-coverage) set(CMAKE_EXE_LINKER_FLAGS "--coverage") @@ -75,7 +76,7 @@ add_compile_options(-DBUILD_VERSION_STRING="${BUILD_VERSION_STRING}") set(ROCBLAS_LIB "rocblas") set(HIP_HCC_LIB "amdhip64") -#ROCBLAS VERSION CHECK FLAGS +#ROCBLAS VERSION CHECK FLAGS TO CHECK REORG VERSION 2.44.0 add_compile_options(-DRVS_ROCBLAS_VERSION_FLAT=${RVS_ROCBLAS_VERSION_FLAT}) # Determine HSA_PATH @@ -93,6 +94,7 @@ set(HIP_HCC_BUILD_FLAGS "${HIP_HCC_BUILD_FLAGS} -DHIP_VERSION_MAJOR=${HIP_VERSIO set(HIP_HCC_BUILD_FLAGS) set(HIP_HCC_BUILD_FLAGS "${HIP_HCC_BUILD_FLAGS} -fPIC ${HCC_CXX_FLAGS} -I${HSA_INC_DIR} ${ASAN_CXX_FLAGS}") + # Set compiler and compiler flags set(CMAKE_CXX_COMPILER "${HIPCC_PATH}/bin/hipcc") set(CMAKE_C_COMPILER "${HIPCC_PATH}/bin/hipcc") @@ -135,28 +137,35 @@ if(DEFINED RVS_ROCMSMI) endif() endif() - if(NOT EXISTS "${HIP_LIB_DIR}/lib${HIP_HCC_LIB}.so") message("ERROR: ROC Runtime libraries can't be found under specified path. Please set HIP_LIB_DIR path. Current value is : " ${HIP_LIB_DIR}) RETURN() endif() +if(DEFINED RVS_AMDSMI) + if(NOT RVS_AMDSMI EQUAL 1) + if(NOT EXISTS "${AMD_SMI_LIB_DIR}/lib${AMD_SMI_LIB}.so") + message("ERROR: ${AMD_SMI_LIB} amd_smi library can't be found!...") + RETURN() + endif() + endif() +endif() + ## define include directories -include_directories(./ ../ ${ROCR_INC_DIR} ${ROCBLAS_INC_DIR} ${HIP_INC_DIR} ${HIPRAND_INC_DIR} ${ROCRAND_INC_DIR} ${HIPBLASLT_INC_DIR} ${HIPBLAS-COMMON_INCLUDE_DIR}) +include_directories(./ ../ ${AMD_SMI_INC_DIR} ${ROCBLAS_INC_DIR} ${ROCR_INC_DIR} ${HIP_INC_DIR} ${HIPRAND_INC_DIR} ${ROCRAND_INC_DIR} ${HIPBLASLT_INC_DIR} ${HIPBLAS-COMMON_INCLUDE_DIR}) # Add directories to look for library files to link -link_directories(${RVS_LIB_DIR} ${ROCR_LIB_DIR} ${ROCBLAS_LIB_DIR} ${ASAN_LIB_PATH} ${HIP_LIB_DIR} ${AMD_SMI_LIB_DIR} ${HIPRAND_LIB_DIR} ${ROCRAND_LIB_DIR}) +link_directories(${RVS_LIB_DIR} ${ROCR_LIB_DIR} ${ROCBLAS_LIB_DIR} ${AMD_SMI_LIB_DIR} ${ASAN_LIB_PATH} ${HIPRAND_LIB_DIR} ${ROCRAND_LIB_DIR} ${HIPBLASLT_LIB_DIR}) ## additional libraries -set (PROJECT_LINK_LIBS rvslib libpthread.so libpci.so libm.so) +set (PROJECT_LINK_LIBS rvslib libpthread.so libm.so) -## define source files -set(SOURCES src/rvs_module.cpp src/action.cpp src/perf_worker.cpp) +set(SOURCES src/rvs_module.cpp src/action.cpp src/pulse_worker.cpp ) ## define target add_library( ${RVS_TARGET} SHARED ${SOURCES}) set_target_properties(${RVS_TARGET} PROPERTIES SUFFIX .so.${LIB_VERSION_STRING} LIBRARY_OUTPUT_DIRECTORY ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}) -target_link_libraries(${RVS_TARGET} ${PROJECT_LINK_LIBS} ${HIP_HCC_LIB} ${ROCBLAS_LIB}) +target_link_libraries(${RVS_TARGET} ${PROJECT_LINK_LIBS} ${HIP_HCC_LIB} ${ROCBLAS_LIB} ${AMD_SMI_LIB}) add_dependencies(${RVS_TARGET} rvslib) add_custom_command(TARGET ${RVS_TARGET} POST_BUILD @@ -175,5 +184,4 @@ if (RVS_BUILD_TESTS) add_custom_command(TARGET ${RVS_TARGET} POST_BUILD COMMAND ln -fs ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/lib${RVS}.so.${VERSION_MAJOR} ${RVS_BINTEST_FOLDER}/lib${RVS}.so WORKING_DIRECTORY ${CMAKE_RUNTIME_OUTPUT_DIRECTORY} ) - include(${CMAKE_CURRENT_SOURCE_DIR}/tests.cmake) endif() diff --git a/pulse.so/include/action.h b/pulse.so/include/action.h new file mode 100644 index 000000000..b61d682e7 --- /dev/null +++ b/pulse.so/include/action.h @@ -0,0 +1,116 @@ +/******************************************************************************** + * + * Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. + * + * MIT LICENSE: + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies + * of the Software, and to permit persons to whom the Software is furnished to do + * so, subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + * + *******************************************************************************/ +#ifndef PULSE_SO_INCLUDE_ACTION_H_ +#define PULSE_SO_INCLUDE_ACTION_H_ + + +#include +#include +#include + +#include "include/rvsactionbase.h" +#include "amd_smi/amdsmi.h" + +using std::vector; +using std::string; + +class PulseWorker; + +/** + * @class pulse_action + * @ingroup PULSE + * + * @brief Pulse test action implementation class + * + * Derives from rvs::actionbase and implements the GPU power pulse + * stress test in its run() method. + */ +class pulse_action: public rvs::actionbase { + public: + pulse_action(); + virtual ~pulse_action(); + virtual int run(void); + + protected: + //! pulse cycle rate in Hz + int pulse_rate; + //! fraction of each cycle spent in high-power phase (0.0-1.0) + float high_phase_ratio; + //! GEMM operation type (sgemm, dgemm, hgemm, etc.) + std::string pulse_ops_type; + //! GEMM data type + std::string pulse_data_type; + //! GEMM output data type + std::string pulse_out_data_type; + //! matrix size for GEMM K dimension + uint64_t pulse_matrix_size; + //! GEMM alpha scalar + float pulse_alpha_val; + //! GEMM beta scalar + float pulse_beta_val; + //! transpose A setting + int pulse_trans_a; + //! transpose B setting + int pulse_trans_b; + //! leading dimension offsets + int pulse_lda_offset; + int pulse_ldb_offset; + int pulse_ldc_offset; + int pulse_ldd_offset; + //! matrix initialization method + std::string pulse_matrix_init; + //! power tolerance percentage + float pulse_tolerance; + //! sampling rate for power readings (ms) + uint64_t pulse_sample_interval; + //! kernel calls between health checks + int pulse_workload_iterations; + //! stop immediately on first error + bool pulse_halt_on_error; + //! cross-GPU sync timeout (ms) + int pulse_gpu_sync_wait; + //! compute verification mode: "crc" or "diff" + std::string pulse_verify_mode; + //! hot calls for BLAS warmup + uint64_t pulse_hot_calls; + //! blas backend source library + std::string pulse_blas_source; + //! gemm compute type + std::string pulse_compute_type; + //! fail high phase if junction/edge temp (C) exceeds this; 0 disables check + float pulse_max_temp_c; + + friend class PulseWorker; + + std::map hip_to_smi_idxs; + void hip_to_smi_indices(); + bool get_all_pulse_config_keys(void); + int get_num_amd_gpu_devices(void); + int get_all_selected_gpus(void); + bool do_pulse_test(std::map pulse_gpus_device_index, + std::vector& mcm_type); +}; + +#endif // PULSE_SO_INCLUDE_ACTION_H_ diff --git a/pulse.so/include/pulse_worker.h b/pulse.so/include/pulse_worker.h new file mode 100644 index 000000000..ccbc3c534 --- /dev/null +++ b/pulse.so/include/pulse_worker.h @@ -0,0 +1,253 @@ +/******************************************************************************** + * + * Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. + * + * MIT LICENSE: + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies + * of the Software, and to permit persons to whom the Software is furnished to do + * so, subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + * + *******************************************************************************/ +#ifndef PULSE_SO_INCLUDE_PULSE_WORKER_H_ +#define PULSE_SO_INCLUDE_PULSE_WORKER_H_ + +#include +#include +#include +#include +#include +#include +#include + +#include "include/rvsthreadbase.h" +#include "include/rvs_blas.h" +#include "include/rvs_util.h" +#include "include/rvsactionbase.h" +#include "include/action.h" + +/** + * @class PulseWorker + * @ingroup PULSE + * + * @brief PulseWorker action implementation class + * + * Derives from rvs::ThreadBase and implements the per-GPU pulse + * stress workload including cross-GPU synchronized avalanche release. + */ +class PulseWorker : public rvs::ThreadBase { + public: + PulseWorker(); + virtual ~PulseWorker(); + + void set_name(const std::string& name) { action_name = name; } + void set_action(const pulse_action& _action) { action = _action; } + const std::string& get_name(void) { return action_name; } + + void set_gpu_id(uint16_t _gpu_id) { gpu_id = _gpu_id; } + uint16_t get_gpu_id(void) { return gpu_id; } + + void set_gpu_device_index(int _gpu_device_index) { + gpu_device_index = _gpu_device_index; + } + int get_gpu_device_index(void) { return gpu_device_index; } + + void set_smi_device_handle(amdsmi_processor_handle _handle) { + smi_device_handle = _handle; + } + amdsmi_processor_handle get_smi_device_handle(void) { + return smi_device_handle; + } + + void set_run_duration_ms(uint64_t _run_duration_ms) { + run_duration_ms = _run_duration_ms; + } + + void set_sample_interval(uint64_t _sample_interval) { + sample_interval = _sample_interval; + } + + void set_log_interval(uint64_t _log_interval) { + log_interval = _log_interval; + } + + void set_pulse_rate(int _pulse_rate) { + pulse_rate = _pulse_rate; + } + + void set_high_phase_ratio(float _ratio) { + high_phase_ratio = _ratio; + } + + void set_tolerance(float _tolerance) { + tolerance = _tolerance; + } + + void set_max_temp_c(float _c) { + max_temp_c = _c; + } + + void set_matrix_size(uint64_t _matrix_size) { + matrix_size = _matrix_size; + } + + void set_ops_type(std::string _ops_type) { + pulse_ops_type = _ops_type; + } + + void set_data_type(std::string _data_type) { + pulse_data_type = _data_type; + } + + void set_out_data_type(std::string _out_data_type) { + pulse_out_data_type = _out_data_type; + } + + void set_matrix_transpose_a(int transa) { pulse_trans_a = transa; } + void set_matrix_transpose_b(int transb) { pulse_trans_b = transb; } + void set_alpha_val(float alpha) { pulse_alpha_val = alpha; } + void set_beta_val(float beta) { pulse_beta_val = beta; } + void set_lda_offset(int lda) { pulse_lda_offset = lda; } + void set_ldb_offset(int ldb) { pulse_ldb_offset = ldb; } + void set_ldc_offset(int ldc) { pulse_ldc_offset = ldc; } + void set_ldd_offset(int ldd) { pulse_ldd_offset = ldd; } + + void set_workload_iterations(int _iters) { + workload_iterations = _iters; + } + + void set_halt_on_error(bool _halt) { + halt_on_error = _halt; + } + + void set_verify_mode(std::string _mode) { + verify_mode = _mode; + } + + void set_hot_calls(uint64_t _hot_calls) { + pulse_hot_calls = _hot_calls; + } + + void set_matrix_init(std::string _init) { + matrix_init = _init; + } + + void set_blas_source(std::string _source) { + blas_source = _source; + } + + void set_compute_type(std::string _type) { + compute_type = _type; + } + + void set_mcm_type(mcm_type_t _mcm_type) { + mcm_type = _mcm_type; + } + + void set_num_gpus(int _num_gpus) { + num_gpus = _num_gpus; + } + + void set_worker_index(int _idx) { + worker_index = _idx; + } + + void set_sync_resources(std::barrier<>* _cpu_barrier, + int32_t* _gpu_arrival_count, + int32_t* _gpu_release_flag, + std::atomic* _done_flag) { + cpu_barrier = _cpu_barrier; + gpu_arrival_count = _gpu_arrival_count; + gpu_release_flag = _gpu_release_flag; + done_flag = _done_flag; + } + + static void set_use_json(bool _bjson) { bjson = _bjson; } + static bool get_use_json(void) { return bjson; } + bool get_result(void) { return result; } + + protected: + virtual void run(void); + bool do_pulse_stress(void); + bool setup_blas(void); + float read_power(void); + float read_temperature(void); + + bool discover_valid_clock_levels(void); + bool set_highest_clocks(void); + bool set_lowest_clocks(void); + bool restore_clocks(void); + + bool gpu_barrier_sync(bool time_up, bool& test_passed); + + bool run_gemm_verify(bool& test_passed); + + protected: + std::unique_ptr gpu_blas; + + std::string action_name; + pulse_action action; + int gpu_device_index; + amdsmi_processor_handle smi_device_handle; + uint16_t gpu_id; + + uint64_t run_duration_ms; + uint64_t sample_interval; + uint64_t log_interval; + int pulse_rate; + float high_phase_ratio; + float tolerance; + float max_temp_c; + uint64_t matrix_size; + + std::string pulse_ops_type; + std::string pulse_data_type; + std::string pulse_out_data_type; + int pulse_trans_a; + int pulse_trans_b; + float pulse_alpha_val; + float pulse_beta_val; + int pulse_lda_offset; + int pulse_ldb_offset; + int pulse_ldc_offset; + int pulse_ldd_offset; + + int workload_iterations; + bool halt_on_error; + std::string verify_mode; + uint64_t pulse_hot_calls; + std::string matrix_init; + std::string blas_source; + std::string compute_type; + mcm_type_t mcm_type; + int num_gpus; + int worker_index; + + //! Cross-GPU synchronization (set by action when parallel + multi-GPU) + std::barrier<>* cpu_barrier; + int32_t* gpu_arrival_count; + int32_t* gpu_release_flag; + std::atomic* done_flag; + + //! Valid clock levels discovered via AMDSMI + std::vector valid_gfx_levels; + std::vector valid_mem_levels; + + static bool bjson; + bool result; +}; + +#endif // PULSE_SO_INCLUDE_PULSE_WORKER_H_ diff --git a/edp.so/include/rvs_module.h b/pulse.so/include/rvs_module.h similarity index 87% rename from edp.so/include/rvs_module.h rename to pulse.so/include/rvs_module.h index 3763a68d7..4d880b21f 100644 --- a/edp.so/include/rvs_module.h +++ b/pulse.so/include/rvs_module.h @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -22,10 +22,9 @@ * SOFTWARE. * *******************************************************************************/ -#ifndef GST_SO_INCLUDE_RVS_MODULE_H_ -#define GST_SO_INCLUDE_RVS_MODULE_H_ +#ifndef PULSE_SO_INCLUDE_RVS_MODULE_H_ +#define PULSE_SO_INCLUDE_RVS_MODULE_H_ #include "include/rvsliblog.h" - -#endif // GST_SO_INCLUDE_RVS_MODULE_H_ +#endif // PULSE_SO_INCLUDE_RVS_MODULE_H_ diff --git a/pulse.so/src/action.cpp b/pulse.so/src/action.cpp new file mode 100644 index 000000000..07f7affac --- /dev/null +++ b/pulse.so/src/action.cpp @@ -0,0 +1,636 @@ +/******************************************************************************** + * + * Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. + * + * MIT LICENSE: + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies + * of the Software, and to permit persons to whom the Software is furnished to do + * so, subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + * + *******************************************************************************/ +#include "include/action.h" + +#include +#include +#include +#include +#include +#include + + +#define __HIP_PLATFORM_HCC__ +#include "hip/hip_runtime.h" +#include "hip/hip_runtime_api.h" + +#include "include/rvs_key_def.h" +#include "include/pulse_worker.h" +#include "include/gpu_util.h" +#include "include/rvs_util.h" +#include "include/rvs_module.h" +#include "include/rvsactionbase.h" +#include "include/rvsloglp.h" +#include "include/rsmi_util.h" + +using std::string; +using std::vector; +using std::map; + +#define RVS_CONF_PULSE_RATE_KEY "pulse_rate" +#define RVS_CONF_HIGH_PHASE_RATIO_KEY "high_phase_ratio" +#define RVS_CONF_TOLERANCE_KEY "tolerance" +#define RVS_CONF_MATRIX_SIZE_KEY "matrix_size" +#define RVS_CONF_OPS_TYPE_KEY "ops_type" +#define RVS_CONF_DATA_TYPE_KEY "data_type" +#define RVS_CONF_OUT_DATA_TYPE_KEY "out_data_type" +#define RVS_CONF_TRANS_A_KEY "transa" +#define RVS_CONF_TRANS_B_KEY "transb" +#define RVS_CONF_ALPHA_VAL_KEY "alpha" +#define RVS_CONF_BETA_VAL_KEY "beta" +#define RVS_CONF_LDA_OFFSET_KEY "lda" +#define RVS_CONF_LDB_OFFSET_KEY "ldb" +#define RVS_CONF_LDC_OFFSET_KEY "ldc" +#define RVS_CONF_LDD_OFFSET_KEY "ldd" +#define RVS_CONF_WORKLOAD_ITERS_KEY "workload_iterations" +#define RVS_CONF_HALT_ON_ERROR_KEY "halt_on_error" +#define RVS_CONF_GPU_SYNC_WAIT_KEY "gpu_sync_wait" +#define RVS_CONF_VERIFY_MODE_KEY "verify_mode" +#define RVS_CONF_HOT_CALLS_KEY "hot_calls" +#define RVS_CONF_MATRIX_INIT_KEY "matrix_init" +#define RVS_CONF_BLAS_SOURCE_KEY "blas_source" +#define RVS_CONF_COMPUTE_TYPE_KEY "compute_type" +#define RVS_CONF_MAX_TEMP_C_KEY "max_temp_c" + +#define PULSE_DEFAULT_RATE 2 +#define PULSE_DEFAULT_HIGH_PHASE_RATIO 0.5f +#define PULSE_DEFAULT_TOLERANCE 10.0f +#define PULSE_DEFAULT_MATRIX_SIZE 4096 +#define PULSE_DEFAULT_OPS_TYPE "sgemm" +#define PULSE_DEFAULT_DATA_TYPE "" +#define PULSE_DEFAULT_OUT_DATA_TYPE "" +#define PULSE_DEFAULT_TRANS_A 0 +#define PULSE_DEFAULT_TRANS_B 1 +#define PULSE_DEFAULT_ALPHA_VAL 2.0f +#define PULSE_DEFAULT_BETA_VAL -1.0f +#define PULSE_DEFAULT_LDA_OFFSET 0 +#define PULSE_DEFAULT_LDB_OFFSET 0 +#define PULSE_DEFAULT_LDC_OFFSET 0 +#define PULSE_DEFAULT_LDD_OFFSET 0 +#define PULSE_DEFAULT_WORKLOAD_ITERS 128 +#define PULSE_DEFAULT_HALT_ON_ERROR false +#define PULSE_DEFAULT_GPU_SYNC_WAIT 10000 +#define PULSE_DEFAULT_VERIFY_MODE "diff" +#define PULSE_DEFAULT_SAMPLE_INTERVAL 100 +#define PULSE_DEFAULT_HOT_CALLS 1 +#define PULSE_DEFAULT_MATRIX_INIT "default" +#define PULSE_DEFAULT_BLAS_SOURCE "rocblas" +#define PULSE_DEFAULT_COMPUTE_TYPE "fp32_r" +#define PULSE_DEFAULT_MAX_TEMP_C 105.0f + +#define PULSE_NO_COMPATIBLE_GPUS "No AMD compatible GPU found!" +#define JSON_CREATE_NODE_ERROR "JSON cannot create node" + +static constexpr auto MODULE_NAME = "pulse"; +static constexpr auto MODULE_NAME_CAPS = "PULSE"; + +// Beta banner shown once at the start of every pulse action invocation. +// Plain ASCII so it survives non-UTF-8 terminals, log scrapers, and CI capture. +static const char* kPulseBetaBanner = +"\n" +"##############################################################################\n" +"# #\n" +"# *** PULSE STRESS TEST - BETA VERSION *** #\n" +"# #\n" +"# This pulse test is a BETA version and is NOT to be used in #\n" +"# production environments. Pass/fail criteria are still being tuned. #\n" +"# #\n" +"##############################################################################\n"; + +pulse_action::pulse_action() { + module_name = MODULE_NAME; + pulse_max_temp_c = PULSE_DEFAULT_MAX_TEMP_C; +} + +pulse_action::~pulse_action() { + property.clear(); +} + +bool pulse_action::get_all_pulse_config_keys(void) { + int error; + string msg; + bool bsts = true; + + if (property_get_int(RVS_CONF_PULSE_RATE_KEY, + &pulse_rate, PULSE_DEFAULT_RATE)) { + msg = "invalid '" + std::string(RVS_CONF_PULSE_RATE_KEY) + + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + if (property_get(RVS_CONF_HIGH_PHASE_RATIO_KEY, + &high_phase_ratio, PULSE_DEFAULT_HIGH_PHASE_RATIO)) { + msg = "invalid '" + std::string(RVS_CONF_HIGH_PHASE_RATIO_KEY) + + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + if (property_get(RVS_CONF_TOLERANCE_KEY, + &pulse_tolerance, PULSE_DEFAULT_TOLERANCE)) { + msg = "invalid '" + std::string(RVS_CONF_TOLERANCE_KEY) + + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + if (property_get_int(RVS_CONF_MATRIX_SIZE_KEY, + &pulse_matrix_size, (uint64_t)PULSE_DEFAULT_MATRIX_SIZE)) { + msg = "invalid '" + std::string(RVS_CONF_MATRIX_SIZE_KEY) + + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + if (property_get(RVS_CONF_OPS_TYPE_KEY, + &pulse_ops_type, std::string(PULSE_DEFAULT_OPS_TYPE))) { + msg = "invalid '" + std::string(RVS_CONF_OPS_TYPE_KEY) + + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + if (property_get(RVS_CONF_DATA_TYPE_KEY, + &pulse_data_type, std::string(PULSE_DEFAULT_DATA_TYPE))) { + msg = "invalid '" + std::string(RVS_CONF_DATA_TYPE_KEY) + + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + if (property_get(RVS_CONF_OUT_DATA_TYPE_KEY, + &pulse_out_data_type, std::string(PULSE_DEFAULT_OUT_DATA_TYPE))) { + msg = "invalid '" + std::string(RVS_CONF_OUT_DATA_TYPE_KEY) + + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + error = property_get_int(RVS_CONF_TRANS_A_KEY, + &pulse_trans_a, PULSE_DEFAULT_TRANS_A); + if (error == 1) { + msg = "invalid '" + std::string(RVS_CONF_TRANS_A_KEY) + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + error = property_get_int(RVS_CONF_TRANS_B_KEY, + &pulse_trans_b, PULSE_DEFAULT_TRANS_B); + if (error == 1) { + msg = "invalid '" + std::string(RVS_CONF_TRANS_B_KEY) + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + error = property_get(RVS_CONF_ALPHA_VAL_KEY, + &pulse_alpha_val, PULSE_DEFAULT_ALPHA_VAL); + if (error == 1) { + msg = "invalid '" + std::string(RVS_CONF_ALPHA_VAL_KEY) + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + error = property_get(RVS_CONF_BETA_VAL_KEY, + &pulse_beta_val, PULSE_DEFAULT_BETA_VAL); + if (error == 1) { + msg = "invalid '" + std::string(RVS_CONF_BETA_VAL_KEY) + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + error = property_get_int(RVS_CONF_LDA_OFFSET_KEY, + &pulse_lda_offset, PULSE_DEFAULT_LDA_OFFSET); + if (error == 1) { + msg = "invalid '" + std::string(RVS_CONF_LDA_OFFSET_KEY) + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + error = property_get_int(RVS_CONF_LDB_OFFSET_KEY, + &pulse_ldb_offset, PULSE_DEFAULT_LDB_OFFSET); + if (error == 1) { + msg = "invalid '" + std::string(RVS_CONF_LDB_OFFSET_KEY) + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + error = property_get_int(RVS_CONF_LDC_OFFSET_KEY, + &pulse_ldc_offset, PULSE_DEFAULT_LDC_OFFSET); + if (error == 1) { + msg = "invalid '" + std::string(RVS_CONF_LDC_OFFSET_KEY) + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + error = property_get_int(RVS_CONF_LDD_OFFSET_KEY, + &pulse_ldd_offset, PULSE_DEFAULT_LDD_OFFSET); + if (error == 1) { + msg = "invalid '" + std::string(RVS_CONF_LDD_OFFSET_KEY) + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + if (property_get_int(RVS_CONF_WORKLOAD_ITERS_KEY, + &pulse_workload_iterations, PULSE_DEFAULT_WORKLOAD_ITERS)) { + msg = "invalid '" + std::string(RVS_CONF_WORKLOAD_ITERS_KEY) + + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + error = property_get(RVS_CONF_HALT_ON_ERROR_KEY, + &pulse_halt_on_error, PULSE_DEFAULT_HALT_ON_ERROR); + if (error == 1) { + msg = "invalid '" + std::string(RVS_CONF_HALT_ON_ERROR_KEY) + + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + if (property_get_int(RVS_CONF_GPU_SYNC_WAIT_KEY, + &pulse_gpu_sync_wait, PULSE_DEFAULT_GPU_SYNC_WAIT)) { + msg = "invalid '" + std::string(RVS_CONF_GPU_SYNC_WAIT_KEY) + + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + if (property_get(RVS_CONF_VERIFY_MODE_KEY, + &pulse_verify_mode, std::string(PULSE_DEFAULT_VERIFY_MODE))) { + msg = "invalid '" + std::string(RVS_CONF_VERIFY_MODE_KEY) + + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + if (property_get_int(RVS_CONF_SAMPLE_INTERVAL_KEY, + &pulse_sample_interval, (uint64_t)PULSE_DEFAULT_SAMPLE_INTERVAL)) { + msg = "invalid '" + std::string(RVS_CONF_SAMPLE_INTERVAL_KEY) + + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + if (property_get_int(RVS_CONF_HOT_CALLS_KEY, + &pulse_hot_calls, (uint64_t)PULSE_DEFAULT_HOT_CALLS)) { + msg = "invalid '" + std::string(RVS_CONF_HOT_CALLS_KEY) + + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + if (property_get(RVS_CONF_MATRIX_INIT_KEY, + &pulse_matrix_init, std::string(PULSE_DEFAULT_MATRIX_INIT))) { + msg = "invalid '" + std::string(RVS_CONF_MATRIX_INIT_KEY) + + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + if (property_get(RVS_CONF_BLAS_SOURCE_KEY, + &pulse_blas_source, std::string(PULSE_DEFAULT_BLAS_SOURCE))) { + msg = "invalid '" + std::string(RVS_CONF_BLAS_SOURCE_KEY) + + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + if (property_get(RVS_CONF_COMPUTE_TYPE_KEY, + &pulse_compute_type, std::string(PULSE_DEFAULT_COMPUTE_TYPE))) { + msg = "invalid '" + std::string(RVS_CONF_COMPUTE_TYPE_KEY) + + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + if (property_get(RVS_CONF_MAX_TEMP_C_KEY, + &pulse_max_temp_c, PULSE_DEFAULT_MAX_TEMP_C)) { + msg = "invalid '" + std::string(RVS_CONF_MAX_TEMP_C_KEY) + + "' key value"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + if (pulse_max_temp_c < 0.0f || + (pulse_max_temp_c > 0.0f && pulse_max_temp_c > 200.0f)) { + msg = "'" + std::string(RVS_CONF_MAX_TEMP_C_KEY) + + "' must be 0 to disable the thermal check, or a limit in (0, 200] °C"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + if (pulse_blas_source != "rocblas" && pulse_blas_source != "hipblaslt") { + msg = "'" + std::string(RVS_CONF_BLAS_SOURCE_KEY) + + "' must be 'rocblas' or 'hipblaslt'"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + // hipBLASLt needs explicit matrix data types; rocBLAS can rely on ops_type alone. + if (pulse_blas_source == "hipblaslt" && pulse_data_type.empty()) { + if (pulse_ops_type == "sgemm") { + pulse_data_type = "fp32_r"; + } else if (pulse_ops_type == "dgemm") { + pulse_data_type = "fp64_r"; + } else if (pulse_ops_type == "hgemm") { + pulse_data_type = "fp16_r"; + } else { + msg = "hipblaslt requires 'data_type' in the action when 'ops_type' is not " + "sgemm, dgemm, or hgemm"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + } + + // Default compute_type is fp32_r; double GEMM with hipBLASLt needs fp64 compute. + if (pulse_blas_source == "hipblaslt" && pulse_ops_type == "dgemm" && + pulse_compute_type == std::string(PULSE_DEFAULT_COMPUTE_TYPE)) { + pulse_compute_type = "fp64_r"; + } + + if (high_phase_ratio < 0.0f || high_phase_ratio > 1.0f) { + msg = "'" + std::string(RVS_CONF_HIGH_PHASE_RATIO_KEY) + + "' must be between 0.0 and 1.0"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + if (pulse_rate <= 0) { + msg = "'" + std::string(RVS_CONF_PULSE_RATE_KEY) + + "' must be positive"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + bsts = false; + } + + if (pulse_sample_interval < 50) { + pulse_sample_interval = 50; + } + + return bsts; +} + +void pulse_action::hip_to_smi_indices(void) { + int hip_num_gpu_devices; + hipGetDeviceCount(&hip_num_gpu_devices); + + std::map smi_map; + smi_map = rvs::get_smi_pci_map(); + + for (int i = 0; i < hip_num_gpu_devices; i++) { + unsigned int pDom, pBus, pDev, pFun; + getBDF(i, pDom, pBus, pDev, pFun); + uint64_t hip_dev_location_id = ( ( ((uint64_t)pDom & 0xffff ) << 32) | + (((uint64_t) pBus & 0xff ) << 8) | (((uint64_t)pDev & 0x1f ) << 3)| ((uint64_t)pFun ) ); + + if(smi_map.find(hip_dev_location_id) != smi_map.end()){ + hip_to_smi_idxs.insert({i, smi_map[hip_dev_location_id]}); + } + } +} + +bool pulse_action::do_pulse_test(map pulse_gpus_device_index, + std::vector& mcm_type) { + std::string msg; + unsigned int i = 0; + int gpuId = 0; + + int num_gpus = static_cast(pulse_gpus_device_index.size()); + vector workers(num_gpus); + + // Shared flag: when any GPU's duration expires, it sets this before + // arriving at the barrier so all GPUs see it and exit together. + std::atomic done_flag{false}; + + // Allocate fine-grained coherent system memory for GPU-side barrier + int32_t* gpu_arrival_count = nullptr; + int32_t* gpu_release_flag = nullptr; + + if (property_parallel && num_gpus > 1) { + if (hipHostMalloc(&gpu_arrival_count, sizeof(int32_t), + hipHostMallocCoherent) != hipSuccess) { + msg = "Failed to allocate fine-grained memory for GPU sync barrier"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + return false; + } + if (hipHostMalloc(&gpu_release_flag, sizeof(int32_t), + hipHostMallocCoherent) != hipSuccess) { + msg = "Failed to allocate fine-grained memory for GPU sync barrier"; + rvs::lp::Err(msg, MODULE_NAME_CAPS, action_name); + hipHostFree(gpu_arrival_count); + return false; + } + *gpu_arrival_count = 0; + *gpu_release_flag = 0; + } + + // CPU-side barrier for host thread alignment (C++20) + std::barrier cpu_barrier(num_gpus); + + for (;;) { + map::iterator it; + + if (property_wait != 0) + sleep(property_wait); + + hip_to_smi_indices(); + + PulseWorker::set_use_json(bjson); + + i = 0; + for (it = pulse_gpus_device_index.begin(); + it != pulse_gpus_device_index.end(); ++it) { + if(hip_to_smi_idxs.find(it->first) != hip_to_smi_idxs.end()){ + workers[i].set_smi_device_handle(hip_to_smi_idxs[it->first]); + } else { + workers[i].set_smi_device_handle(nullptr); + msg = "[" + action_name + "] " + MODULE_NAME + " " + + std::to_string(i) + " has no SMI handle"; + rvs::lp::Log(msg, rvs::logerror); + } + gpuId = it->second; + + workers[i].set_name(action_name); + workers[i].set_action(*this); + workers[i].set_gpu_id(it->second); + workers[i].set_gpu_device_index(it->first); + workers[i].set_run_duration_ms(property_duration); + workers[i].set_sample_interval(pulse_sample_interval); + workers[i].set_log_interval(property_log_interval); + workers[i].set_pulse_rate(pulse_rate); + workers[i].set_high_phase_ratio(high_phase_ratio); + workers[i].set_tolerance(pulse_tolerance); + workers[i].set_matrix_size(pulse_matrix_size); + workers[i].set_ops_type(pulse_ops_type); + workers[i].set_data_type(pulse_data_type); + workers[i].set_out_data_type(pulse_out_data_type); + workers[i].set_matrix_transpose_a(pulse_trans_a); + workers[i].set_matrix_transpose_b(pulse_trans_b); + workers[i].set_alpha_val(pulse_alpha_val); + workers[i].set_beta_val(pulse_beta_val); + workers[i].set_lda_offset(pulse_lda_offset); + workers[i].set_ldb_offset(pulse_ldb_offset); + workers[i].set_ldc_offset(pulse_ldc_offset); + workers[i].set_ldd_offset(pulse_ldd_offset); + workers[i].set_workload_iterations(pulse_workload_iterations); + workers[i].set_halt_on_error(pulse_halt_on_error); + workers[i].set_verify_mode(pulse_verify_mode); + workers[i].set_hot_calls(pulse_hot_calls); + workers[i].set_matrix_init(pulse_matrix_init); + workers[i].set_blas_source(pulse_blas_source); + workers[i].set_compute_type(pulse_compute_type); + workers[i].set_max_temp_c(pulse_max_temp_c); + workers[i].set_mcm_type(mcm_type[i]); + workers[i].set_num_gpus(num_gpus); + workers[i].set_worker_index(i); + + if (property_parallel && num_gpus > 1) { + workers[i].set_sync_resources(&cpu_barrier, + gpu_arrival_count, gpu_release_flag, &done_flag); + } + + i++; + } + + if (property_parallel) { + for (i = 0; i < pulse_gpus_device_index.size(); i++) + workers[i].start(); + for (i = 0; i < pulse_gpus_device_index.size(); i++) + workers[i].join(); + } else { + for (i = 0; i < pulse_gpus_device_index.size(); i++) { + workers[i].start(); + workers[i].join(); + if (rvs::lp::Stopping()) { + if (gpu_arrival_count) hipHostFree(gpu_arrival_count); + if (gpu_release_flag) hipHostFree(gpu_release_flag); + return false; + } + } + } + + msg = "[" + action_name + "] " + MODULE_NAME + " " + + std::to_string(gpuId) + " Completed pulse cycle"; + rvs::lp::Log(msg, rvs::loginfo); + + if (rvs::lp::Stopping()) { + if (gpu_arrival_count) hipHostFree(gpu_arrival_count); + if (gpu_release_flag) hipHostFree(gpu_release_flag); + return false; + } + + // single iteration for pulse test + break; + } + + if (gpu_arrival_count) hipHostFree(gpu_arrival_count); + if (gpu_release_flag) hipHostFree(gpu_release_flag); + + for (i = 0; i < pulse_gpus_device_index.size(); i++) { + if(false == workers[i].get_result()) + return false; + } + + return true; +} + +int pulse_action::get_num_amd_gpu_devices(void) { + int hip_num_gpu_devices; + hipGetDeviceCount(&hip_num_gpu_devices); + return hip_num_gpu_devices; +} + +int pulse_action::get_all_selected_gpus(void) { + int hip_num_gpu_devices; + bool amd_gpus_found = false; + map pulse_gpus_device_index; + std::string msg; + std::vector mcm_type; + + hipGetDeviceCount(&hip_num_gpu_devices); + if (hip_num_gpu_devices < 1) + return -1; + + amd_gpus_found = fetch_gpu_list(hip_num_gpu_devices, + pulse_gpus_device_index, + property_device, property_device_id, property_device_all, + property_device_index, property_device_index_all, true, &mcm_type); + if(!amd_gpus_found){ + msg = "No devices match criteria from the test configuration."; + rvs::lp::Log(msg, rvs::logerror); + if (bjson) { + unsigned int sec; + unsigned int usec; + rvs::lp::get_ticks(&sec, &usec); + void *json_root_node = rvs::lp::LogRecordCreate(MODULE_NAME, + action_name.c_str(), rvs::logerror, sec, usec, true); + if (!json_root_node) { + string emsg = std::string(JSON_CREATE_NODE_ERROR); + rvs::lp::Err(emsg, MODULE_NAME_CAPS, action_name); + return -1; + } + rvs::lp::AddString(json_root_node, "ERROR", + "No AMD compatible GPU found!"); + rvs::lp::LogRecordFlush(json_root_node, rvs::logerror); + } + return -1; + } + + int res = 0; + if(do_pulse_test(pulse_gpus_device_index, mcm_type)) + res = 0; + else + res = -1; + return res; +} + +int pulse_action::run(void) { + string msg; + rvs::action_result_t action_result; + + rvs::lp::Log(std::string(kPulseBetaBanner), rvs::logresults); + + if (!get_all_common_config_keys()) + return -1; + + if (!get_all_pulse_config_keys()) + return -1; + + if(bjson){ + json_add_primary_fields(std::string(MODULE_NAME), action_name); + } + + auto res = get_all_selected_gpus(); + if(bjson){ + rvs::lp::JsonActionEndNodeCreate(); + } + + action_result.state = rvs::actionstate::ACTION_COMPLETED; + action_result.status = (!res) ? rvs::actionstatus::ACTION_SUCCESS + : rvs::actionstatus::ACTION_FAILED; + action_result.output = "PULSE Module action " + action_name + " completed"; + action_callback(&action_result); + + return res; +} diff --git a/pulse.so/src/pulse_worker.cpp b/pulse.so/src/pulse_worker.cpp new file mode 100644 index 000000000..4a498b72d --- /dev/null +++ b/pulse.so/src/pulse_worker.cpp @@ -0,0 +1,765 @@ +/******************************************************************************** + * + * Pulse worker: alternating GEMM / idle phases, SMI power & temperature, + * optional end-of-run GEMM verify (verify_mode / tolerance; CPU accuracy + * skipped for matrix_size > 2048), sample_interval throttling, multi-GPU + * barriers on the default HIP stream, configurable max_temp_c. + * + * Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. + * + * MIT LICENSE: + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to + * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies + * of the Software, and to permit persons to whom the Software is furnished to do + * so, subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + * SOFTWARE. + * + *******************************************************************************/ + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "hip/hip_runtime.h" +#include "hip/hip_runtime_api.h" +#include "include/rvs_module.h" +#include "include/rvsloglp.h" +#include "include/pulse_worker.h" + +#define MODULE_NAME "pulse" +#define PULSE_PASS_KEY "pass" +#define PULSE_RESULT_PASS "TRUE" +#define PULSE_RESULT_FAIL "FALSE" +#define PULSE_JSON_LOG_GPU_ID_KEY "gpu_id" +#define PULSE_JSON_POWER_HIGH_KEY "power_high" +#define PULSE_JSON_POWER_LOW_KEY "power_low" + +using std::string; + +bool PulseWorker::bjson = false; + +static string pulse_mode_normalize(const string& in) { + string out; + for (char c : in) { + if (!std::isspace(static_cast(c))) + out.push_back(static_cast( + std::tolower(static_cast(c)))); + } + return out; +} + +/** Map verify_mode to rvs_blas::validate_gemm flags (GST-style semantics). */ +static void pulse_verify_flags(const string& verify_mode, + const string& ops_type, const string& data_type, + bool& self_check, bool& accu_check) { + self_check = false; + accu_check = false; + const string m = pulse_mode_normalize(verify_mode); + if (m.empty() || m == "none" || m == "off" || m == "false") + return; + if (m == "crc") { + self_check = true; + return; + } + if (m == "diff") { + if (ops_type == "sgemm" || ops_type == "dgemm") { + accu_check = true; + } else if (data_type == "fp16_r" || data_type == "bf16_r" || + data_type == "fp8_r" || ops_type == "hgemm") { + self_check = true; + } else { + self_check = true; + } + return; + } + if (m == "both" || m == "full") { + self_check = true; + if (ops_type == "sgemm" || ops_type == "dgemm") + accu_check = true; + return; + } +} + +static uint64_t time_diff( + std::chrono::time_point t_end, + std::chrono::time_point t_start) { + auto milliseconds = std::chrono::duration_cast( + t_end - t_start); + return milliseconds.count(); +} + +// GPU-side barrier kernel for synchronized avalanche release. +// Uses fine-grained coherent system memory allocated via hipHostMallocCoherent. +// All GPUs atomically signal arrival, then spin until the last GPU releases. +__global__ void gpu_sync_barrier_kernel(int32_t* arrival_count, + int32_t* release_flag, + int32_t target_count) { + int32_t arrived = atomicAdd_system(arrival_count, 1) + 1; + + if (arrived == target_count) { + __atomic_store_n(release_flag, 1, __ATOMIC_RELEASE); + } else { + while (__atomic_load_n(release_flag, __ATOMIC_ACQUIRE) == 0) { + // tight spin on GPU hardware — nanosecond resolution + } + } +} + +PulseWorker::PulseWorker() + : gpu_device_index(-1), + smi_device_handle(nullptr), + gpu_id(0), + run_duration_ms(0), + sample_interval(100), + log_interval(1000), + pulse_rate(2), + high_phase_ratio(0.5f), + tolerance(10.0f), + max_temp_c(105.0f), + matrix_size(4096), + pulse_trans_a(0), + pulse_trans_b(1), + pulse_alpha_val(2.0f), + pulse_beta_val(-1.0f), + pulse_lda_offset(0), + pulse_ldb_offset(0), + pulse_ldc_offset(0), + pulse_ldd_offset(0), + workload_iterations(128), + halt_on_error(false), + pulse_hot_calls(1), + num_gpus(1), + worker_index(0), + cpu_barrier(nullptr), + gpu_arrival_count(nullptr), + gpu_release_flag(nullptr), + done_flag(nullptr), + result(false) { +} + +PulseWorker::~PulseWorker() { +} + +float PulseWorker::read_power(void) { + amdsmi_power_info_t pwr_info; + amdsmi_status_t stat = amdsmi_get_power_info(smi_device_handle, &pwr_info); + if (stat == AMDSMI_STATUS_SUCCESS) { + return static_cast(pwr_info.socket_power); + } + return -1.0f; +} + +// amdsmi_get_temp_metric: some stacks return millidegree Celsius (e.g. 43000 +// for 43 °C); others return whole degrees in the int64 (as in +// tst_worker). Values with magnitude above 1000 are treated as millidegrees. +static float amdsmi_temperature_to_celsius(int64_t raw) { + const int64_t mag = raw >= 0 ? raw : -raw; + if (mag > 1000) + return static_cast(raw) / 1000.0f; + return static_cast(raw); +} + +float PulseWorker::read_temperature(void) { + int64_t temp = 0; + amdsmi_status_t stat = amdsmi_get_temp_metric(smi_device_handle, + AMDSMI_TEMPERATURE_TYPE_JUNCTION, AMDSMI_TEMP_CURRENT, &temp); + if (stat != AMDSMI_STATUS_SUCCESS) { + stat = amdsmi_get_temp_metric(smi_device_handle, + AMDSMI_TEMPERATURE_TYPE_EDGE, AMDSMI_TEMP_CURRENT, &temp); + } + if (stat == AMDSMI_STATUS_SUCCESS) { + return amdsmi_temperature_to_celsius(temp); + } + return -1.0f; +} + +bool PulseWorker::discover_valid_clock_levels(void) { + amdsmi_frequencies_t freqs{}; + string msg; + + // Discover GFX clock levels — brief sleep after each set_clk_freq so the + // driver can apply the request before the next probe. + if (amdsmi_get_clk_freq(smi_device_handle, AMDSMI_CLK_TYPE_SYS, &freqs) + == AMDSMI_STATUS_SUCCESS) { + for (uint32_t level = 0; level < freqs.num_supported; ++level) { + uint64_t test_mask = (1ULL << level); + if (amdsmi_set_clk_freq(smi_device_handle, AMDSMI_CLK_TYPE_SYS, + test_mask) == AMDSMI_STATUS_SUCCESS) { + valid_gfx_levels.push_back(level); + } else { + break; + } + std::this_thread::sleep_for(std::chrono::milliseconds(20)); + } + // Restore all valid levels after probing + if (!valid_gfx_levels.empty()) { + uint64_t restore = 0; + for (auto lvl : valid_gfx_levels) restore |= (1ULL << lvl); + amdsmi_set_clk_freq(smi_device_handle, AMDSMI_CLK_TYPE_SYS, restore); + } + } + + // Discover MEM clock levels + if (amdsmi_get_clk_freq(smi_device_handle, AMDSMI_CLK_TYPE_MEM, &freqs) + == AMDSMI_STATUS_SUCCESS) { + for (uint32_t level = 0; level < freqs.num_supported; ++level) { + uint64_t test_mask = (1ULL << level); + if (amdsmi_set_clk_freq(smi_device_handle, AMDSMI_CLK_TYPE_MEM, + test_mask) == AMDSMI_STATUS_SUCCESS) { + valid_mem_levels.push_back(level); + } else { + break; + } + std::this_thread::sleep_for(std::chrono::milliseconds(20)); + } + if (!valid_mem_levels.empty()) { + uint64_t restore = 0; + for (auto lvl : valid_mem_levels) restore |= (1ULL << lvl); + amdsmi_set_clk_freq(smi_device_handle, AMDSMI_CLK_TYPE_MEM, restore); + } + } + + msg = "[" + action_name + "] " + MODULE_NAME + " " + + std::to_string(gpu_id) + " discovered " + + std::to_string(valid_gfx_levels.size()) + " GFX levels, " + + std::to_string(valid_mem_levels.size()) + " MEM levels"; + rvs::lp::Log(msg, rvs::loginfo); + + return !valid_gfx_levels.empty(); +} + +bool PulseWorker::set_highest_clocks(void) { + bool ok = true; + amdsmi_frequencies_t freqs{}; + + if (!valid_gfx_levels.empty()) { + if (amdsmi_get_clk_freq(smi_device_handle, AMDSMI_CLK_TYPE_SYS, &freqs) + == AMDSMI_STATUS_SUCCESS) { + uint32_t max_idx = valid_gfx_levels[0]; + for (auto level : valid_gfx_levels) { + if (freqs.frequency[level] > freqs.frequency[max_idx]) + max_idx = level; + } + uint64_t mask = (1ULL << max_idx); + if (amdsmi_set_clk_freq(smi_device_handle, AMDSMI_CLK_TYPE_SYS, mask) + != AMDSMI_STATUS_SUCCESS) { + ok = false; + } + } + } + + if (!valid_mem_levels.empty()) { + if (amdsmi_get_clk_freq(smi_device_handle, AMDSMI_CLK_TYPE_MEM, &freqs) + == AMDSMI_STATUS_SUCCESS) { + uint32_t max_idx = valid_mem_levels[0]; + for (auto level : valid_mem_levels) { + if (freqs.frequency[level] > freqs.frequency[max_idx]) + max_idx = level; + } + uint64_t mask = (1ULL << max_idx); + if (amdsmi_set_clk_freq(smi_device_handle, AMDSMI_CLK_TYPE_MEM, mask) + != AMDSMI_STATUS_SUCCESS) { + ok = false; + } + } + } + + return ok; +} + +bool PulseWorker::set_lowest_clocks(void) { + bool ok = true; + amdsmi_frequencies_t freqs{}; + + if (!valid_gfx_levels.empty()) { + if (amdsmi_get_clk_freq(smi_device_handle, AMDSMI_CLK_TYPE_SYS, &freqs) + == AMDSMI_STATUS_SUCCESS) { + uint32_t min_idx = valid_gfx_levels[0]; + for (auto level : valid_gfx_levels) { + if (freqs.frequency[level] < freqs.frequency[min_idx]) + min_idx = level; + } + uint64_t mask = (1ULL << min_idx); + if (amdsmi_set_clk_freq(smi_device_handle, AMDSMI_CLK_TYPE_SYS, mask) + != AMDSMI_STATUS_SUCCESS) { + ok = false; + } + } + } + + if (!valid_mem_levels.empty()) { + if (amdsmi_get_clk_freq(smi_device_handle, AMDSMI_CLK_TYPE_MEM, &freqs) + == AMDSMI_STATUS_SUCCESS) { + uint32_t min_idx = valid_mem_levels[0]; + for (auto level : valid_mem_levels) { + if (freqs.frequency[level] < freqs.frequency[min_idx]) + min_idx = level; + } + uint64_t mask = (1ULL << min_idx); + if (amdsmi_set_clk_freq(smi_device_handle, AMDSMI_CLK_TYPE_MEM, mask) + != AMDSMI_STATUS_SUCCESS) { + ok = false; + } + } + } + + return ok; +} + +bool PulseWorker::restore_clocks(void) { + bool ok = true; + + if (!valid_gfx_levels.empty()) { + uint64_t mask = 0; + for (auto lvl : valid_gfx_levels) mask |= (1ULL << lvl); + if (amdsmi_set_clk_freq(smi_device_handle, AMDSMI_CLK_TYPE_SYS, mask) + != AMDSMI_STATUS_SUCCESS) { + ok = false; + } + } + + if (!valid_mem_levels.empty()) { + uint64_t mask = 0; + for (auto lvl : valid_mem_levels) mask |= (1ULL << lvl); + if (amdsmi_set_clk_freq(smi_device_handle, AMDSMI_CLK_TYPE_MEM, mask) + != AMDSMI_STATUS_SUCCESS) { + ok = false; + } + } + + return ok; +} + +bool PulseWorker::setup_blas(void) { + int m = static_cast(matrix_size); + int n = static_cast(matrix_size); + int k = static_cast(matrix_size); + + gpu_blas = std::unique_ptr(new rvs_blas( + gpu_device_index, + m, n, k, + matrix_init, + pulse_trans_a, pulse_trans_b, + pulse_alpha_val, pulse_beta_val, + pulse_lda_offset, pulse_ldb_offset, + pulse_ldc_offset, pulse_ldd_offset, + pulse_ops_type, pulse_data_type, + "", 0, + 0, 0, 0, 0, + blas_source, compute_type, + pulse_out_data_type, + "", "", 0, + pulse_hot_calls)); + + gpu_blas->generate_random_matrix_data(); + if (!gpu_blas->copy_data_to_gpu()) { + return false; + } + return true; +} + +bool PulseWorker::gpu_barrier_sync(bool time_up, bool& test_passed) { + if (!cpu_barrier || !gpu_arrival_count || !gpu_release_flag) + return !time_up; + + // Signal done BEFORE barrier so all threads see it atomically + if (time_up && done_flag) + done_flag->store(true, std::memory_order_release); + + // Level 1: CPU barrier — all threads MUST arrive even if done, + // otherwise the remaining threads deadlock here forever. + cpu_barrier->arrive_and_wait(); + + // After barrier, if any thread signaled done, all exit together + if (done_flag && done_flag->load(std::memory_order_acquire)) + return false; + + // Reset sync counters (first worker resets, others wait) + if (worker_index == 0) { + __atomic_store_n(gpu_arrival_count, 0, __ATOMIC_RELEASE); + __atomic_store_n(gpu_release_flag, 0, __ATOMIC_RELEASE); + } + cpu_barrier->arrive_and_wait(); + + if (gpu_device_index >= 0) + hipSetDevice(gpu_device_index); + + // Default stream + device sync (same pattern as pre-stream refactor). + gpu_sync_barrier_kernel<<<1, 1>>>( + gpu_arrival_count, gpu_release_flag, num_gpus); + if (hipDeviceSynchronize() != hipSuccess) { + string msg = "[" + action_name + "] " + MODULE_NAME + " " + + std::to_string(gpu_id) + " hipDeviceSynchronize failed after " + "GPU barrier kernel"; + rvs::lp::Err(msg, "PULSE", action_name); + test_passed = false; + } + return true; +} + +bool PulseWorker::run_gemm_verify(bool& test_passed) { + bool self_check = false; + bool accu_check = false; + pulse_verify_flags(verify_mode, pulse_ops_type, pulse_data_type, + self_check, accu_check); + if (!self_check && !accu_check) + return true; + + // CPU accuracy reference is O(n³) — do not run per pulse on large matrices. + constexpr uint64_t kMaxAccuMatrixDim = 2048ULL; + if (accu_check && matrix_size > kMaxAccuMatrixDim) { + accu_check = false; + if (!self_check) + self_check = true; + } + + if (gpu_device_index >= 0) + hipSetDevice(gpu_device_index); + + double self_error = 0.0; + double accu_error = 0.0; + if (!gpu_blas->validate_gemm(self_check, accu_check, self_error, + accu_error)) { + string msg = "[" + action_name + "] " + MODULE_NAME + " " + + std::to_string(gpu_id) + " GEMM validate_gemm failed (unsupported " + "combination for verify_mode)"; + rvs::lp::Err(msg, "PULSE", action_name); + test_passed = false; + return false; + } + + const double tol = static_cast(tolerance); + if (self_check && self_error > tol) { + string msg = "[" + action_name + "] " + MODULE_NAME + " " + + std::to_string(gpu_id) + " GEMM self-check error " + + std::to_string(self_error) + " exceeds tolerance " + + std::to_string(tol); + rvs::lp::Err(msg, "PULSE", action_name); + test_passed = false; + return false; + } + if (accu_check && accu_error > tol) { + string msg = "[" + action_name + "] " + MODULE_NAME + " " + + std::to_string(gpu_id) + " GEMM accuracy error " + + std::to_string(accu_error) + " exceeds tolerance " + + std::to_string(tol); + rvs::lp::Err(msg, "PULSE", action_name); + test_passed = false; + return false; + } + return true; +} + +bool PulseWorker::do_pulse_stress(void) { + std::chrono::time_point test_start, phase_start, + now, last_log_time, last_sample_time; + string msg; + char gpuid_buff[12]; + rvs::action_result_t action_result; + auto desc = action_descriptor{action_name, MODULE_NAME, gpu_id}; + + snprintf(gpuid_buff, sizeof(gpuid_buff), "%5d", gpu_id); + + if (!setup_blas()) { + msg = "[" + action_name + "] " + MODULE_NAME + " " + + std::to_string(gpu_id) + " BLAS setup failed"; + rvs::lp::Err(msg, "PULSE", action_name); + result = false; + return false; + } + + discover_valid_clock_levels(); + + float max_power_high = 0.0f; + float min_power_low = 999999.0f; + float total_power_high = 0.0f; + float total_power_low = 0.0f; + int high_samples = 0; + int low_samples = 0; + int pulse_count = 0; + bool test_passed = true; + + // Phase durations in milliseconds. + // For visible power deltas use pulse_rate 1-10 Hz (100ms-1s phases). + // Sub-10ms phases are too fast for real GPU power state transitions + // and the SMI reporting window (~100ms averaging). + double period_ms = 1000.0 / static_cast(pulse_rate); + double high_phase_ms = period_ms * high_phase_ratio; + double low_phase_ms = period_ms * (1.0 - high_phase_ratio); + + high_phase_ms = std::max(high_phase_ms, 10.0); + low_phase_ms = std::max(low_phase_ms, 10.0); + + const uint64_t smpl_ms = std::max(1ULL, sample_interval); + + msg = "[" + action_name + "] " + MODULE_NAME + " " + + std::to_string(gpu_id) + " pulse_rate=" + std::to_string(pulse_rate) + + "Hz period=" + std::to_string(period_ms) + + "ms high=" + std::to_string(high_phase_ms) + + "ms low=" + std::to_string(low_phase_ms) + "ms"; + rvs::lp::Log(msg, rvs::loginfo); + + test_start = std::chrono::system_clock::now(); + last_log_time = test_start; + + while (!rvs::lp::Stopping()) { + now = std::chrono::system_clock::now(); + uint64_t elapsed_ms = time_diff(now, test_start); + bool time_up = (run_duration_ms > 0 && elapsed_ms >= run_duration_ms); + + // Coordinated shutdown: when using multi-GPU barrier, all GPUs must + // arrive at the barrier even if their timer expired — otherwise the + // remaining GPUs deadlock. After the barrier, if ANY GPU signaled + // done, ALL GPUs exit together. + if (cpu_barrier && num_gpus > 1) { + if (!gpu_barrier_sync(time_up, test_passed)) + break; + } else if (time_up) { + break; + } + + float pulse_peak_high = 0.0f; + float pulse_trough_low = 999999.0f; + float pulse_temp = -1.0f; + int pulse_gemm_count = 0; + + // ═══ HIGH PHASE: Pin clocks to max + sustained GEMM load ═══ + set_highest_clocks(); + + phase_start = std::chrono::system_clock::now(); + last_sample_time = phase_start; + + while (true) { + now = std::chrono::system_clock::now(); + double phase_elapsed = std::chrono::duration( + now - phase_start).count(); + if (phase_elapsed >= high_phase_ms) + break; + + for (int iter = 0; iter < workload_iterations && !rvs::lp::Stopping(); + ++iter) { + if (!gpu_blas->run_blas_gemm(1)) { + test_passed = false; + if (halt_on_error) goto done; + break; + } + if (!gpu_blas->is_gemm_op_complete()) { + test_passed = false; + if (halt_on_error) goto done; + break; + } + pulse_gemm_count++; + now = std::chrono::system_clock::now(); + if (std::chrono::duration( + now - phase_start).count() >= high_phase_ms) + break; + } + + now = std::chrono::system_clock::now(); + if (time_diff(now, last_sample_time) >= smpl_ms) { + float power = read_power(); + if (power > 0) { + total_power_high += power; + high_samples++; + if (power > max_power_high) max_power_high = power; + if (power > pulse_peak_high) pulse_peak_high = power; + } + + float temp = read_temperature(); + if (temp > 0) pulse_temp = temp; + if (max_temp_c > 0.0f && temp > max_temp_c) { + msg = "[" + action_name + "] " + MODULE_NAME + " " + + std::to_string(gpu_id) + " thermal violation: " + + std::to_string(temp) + "C (limit " + std::to_string(max_temp_c) + + "C)"; + rvs::lp::Log(msg, rvs::logerror); + test_passed = false; + if (halt_on_error) goto done; + } + last_sample_time = now; + } + } + + // Drain the GPU pipeline so all GEMM work finishes before the + // low phase — without this, kernels still in flight keep power + // elevated and the "low" reading is indistinguishable from "high". + hipDeviceSynchronize(); + now = std::chrono::system_clock::now(); + double actual_high_ms = std::chrono::duration( + now - phase_start).count(); + + // ═══ LOW PHASE: Pin clocks to minimum + idle with sampling ═══ + set_lowest_clocks(); + + phase_start = std::chrono::system_clock::now(); + last_sample_time = phase_start; + while (true) { + now = std::chrono::system_clock::now(); + double phase_elapsed = std::chrono::duration( + now - phase_start).count(); + if (phase_elapsed >= low_phase_ms) + break; + + double remain_ms = low_phase_ms - phase_elapsed; + unsigned sleep_chunk = static_cast( + std::min(remain_ms, static_cast(smpl_ms))); + if (sleep_chunk < 1u) + sleep_chunk = 1u; + std::this_thread::sleep_for(std::chrono::milliseconds(sleep_chunk)); + + now = std::chrono::system_clock::now(); + if (time_diff(now, last_sample_time) >= smpl_ms) { + float power_low = read_power(); + if (power_low > 0) { + total_power_low += power_low; + low_samples++; + if (power_low < min_power_low) min_power_low = power_low; + if (power_low < pulse_trough_low) pulse_trough_low = power_low; + } + last_sample_time = now; + } + } + + now = std::chrono::system_clock::now(); + double actual_low_ms = std::chrono::duration( + now - phase_start).count(); + + pulse_count++; + + if (PulseWorker::bjson) { + uint64_t pulse_elapsed_ms = time_diff(now, test_start); + log_to_json(desc, rvs::logresults, + "record_type", "pulse", + "pulse_num", std::to_string(pulse_count), + "elapsed_ms", std::to_string(pulse_elapsed_ms), + "power_high_w", std::to_string(pulse_peak_high), + "power_low_w", std::to_string( + pulse_trough_low < 999999.0f ? pulse_trough_low : 0.0f), + "power_delta_w", std::to_string( + pulse_peak_high - (pulse_trough_low < 999999.0f + ? pulse_trough_low : 0.0f)), + "high_duration_ms", std::to_string(actual_high_ms), + "low_duration_ms", std::to_string(actual_low_ms), + "temp_c", std::to_string(pulse_temp), + "gemm_count", std::to_string(pulse_gemm_count)); + } + if (time_diff(now, last_log_time) >= log_interval) { + float avg_high = (high_samples > 0) + ? total_power_high / high_samples : 0.0f; + float avg_low = (low_samples > 0) + ? total_power_low / low_samples : 0.0f; + msg = "[" + action_name + "] [GPU:: " + gpuid_buff + "] " + + "pulse #" + std::to_string(pulse_count) + + " avg_high=" + std::to_string(avg_high) + "W" + + " avg_low=" + std::to_string(avg_low) + "W" + + " max_high=" + std::to_string(max_power_high) + "W" + + " min_low=" + std::to_string(min_power_low) + "W" + + " delta=" + std::to_string(avg_high - avg_low) + "W"; + rvs::lp::Log(msg, rvs::logresults); + last_log_time = now; + } + + if (rvs::lp::Stopping()) + break; + } + +done: + if (gpu_blas && gpu_device_index >= 0) { + hipSetDevice(gpu_device_index); + hipDeviceSynchronize(); + (void)run_gemm_verify(test_passed); + } + restore_clocks(); + + float avg_power_high = (high_samples > 0) + ? total_power_high / high_samples : 0.0f; + float avg_power_low = (low_samples > 0) + ? total_power_low / low_samples : 0.0f; + float power_delta = avg_power_high - avg_power_low; + + msg = "[" + action_name + "] [GPU:: " + gpuid_buff + "] " + + "completed " + std::to_string(pulse_count) + " pulses" + + " avg_high=" + std::to_string(avg_power_high) + "W" + + " avg_low=" + std::to_string(avg_power_low) + "W" + + " delta=" + std::to_string(power_delta) + "W" + + " max_high=" + std::to_string(max_power_high) + "W" + + " min_low=" + std::to_string(min_power_low) + "W"; + rvs::lp::Log(msg, rvs::logresults); + + if (mcm_type == mcm_type_t::SECONDARY) { + msg = "[" + action_name + "] " + MODULE_NAME + " " + + std::to_string(gpu_id) + + " secondary MCM, considering pass by default"; + rvs::lp::Log(msg, rvs::loginfo); + result = true; + } else { + result = test_passed && (pulse_count > 0); + } + + if (PulseWorker::bjson) { + log_to_json(desc, rvs::logresults, + "record_type", "summary", + "pulse_count", std::to_string(pulse_count), + "avg_power_high_w", std::to_string(avg_power_high), + "avg_power_low_w", std::to_string(avg_power_low), + "avg_delta_w", std::to_string(power_delta), + "max_power_high_w", std::to_string(max_power_high), + "min_power_low_w", std::to_string(min_power_low), + "duration_ms", std::to_string(run_duration_ms), + "pulse_rate_hz", std::to_string(pulse_rate), + "ops_type", pulse_ops_type, + "matrix_size", std::to_string(matrix_size), + PULSE_PASS_KEY, result ? "true" : "false"); + } + + action_result.state = rvs::actionstate::ACTION_RUNNING; + action_result.status = result + ? rvs::actionstatus::ACTION_SUCCESS + : rvs::actionstatus::ACTION_FAILED; + action_result.output = msg; + action.action_callback(&action_result); + + return result; +} + +void PulseWorker::run() { + string msg; + char gpuid_buff[12]; + + msg = "[" + action_name + "] " + MODULE_NAME + " " + + std::to_string(gpu_id) + " start pulse_rate=" + + std::to_string(pulse_rate); + rvs::lp::Log(msg, rvs::loginfo); + + bool pass = do_pulse_stress(); + + if (rvs::lp::Stopping()) + return; + + snprintf(gpuid_buff, sizeof(gpuid_buff), "%5d", gpu_id); + msg = "[" + action_name + "] [GPU:: " + gpuid_buff + "] " + + PULSE_PASS_KEY + ": " + + (pass ? PULSE_RESULT_PASS : PULSE_RESULT_FAIL); + rvs::lp::Log(msg, rvs::logresults); + + sleep(2); +} diff --git a/perf.so/src/rvs_module.cpp b/pulse.so/src/rvs_module.cpp similarity index 71% rename from perf.so/src/rvs_module.cpp rename to pulse.so/src/rvs_module.cpp index d3bb4e136..f323a35fa 100644 --- a/perf.so/src/rvs_module.cpp +++ b/pulse.so/src/rvs_module.cpp @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -28,16 +28,16 @@ #include "include/gpu_util.h" /** - * @defgroup PERF PERF Module + * @defgroup PULSE PULSE Module * - * @brief performs GPU Stress Test + * @brief GPU Power Pulse Stress Test + * + * The Pulse Test fluctuates GPU power usage by alternating between + * high-compute and idle phases at configurable rates, creating rapid + * current spikes that stress the power supply and voltage regulators. + * Work across GPUs is synchronized via a two-level barrier (CPU + GPU + * fine-grained atomics) to create maximum aggregate current transients. * - * The GPU Stress Test runs a Graphics Stress test or SGEMM/DGEMM - * (Single/Double-precision General Matrix Multiplication) workload - * on one, some or all GPUs. The GPUs can be of the same or different types. - * The duration of the benchmark should be configurable, both in terms of time - * (how long to run) and iterations (how many times to run). - * */ extern "C" int rvs_module_has_interface(int iid) { @@ -51,14 +51,17 @@ extern "C" int rvs_module_has_interface(int iid) { } extern "C" const char* rvs_module_get_description(void) { - return "ROCm Validation Suite PERF module"; + return "GPU Power Pulse Stress Test - creates rapid power fluctuations " + "to stress the GPU power supply and voltage regulators."; } extern "C" const char* rvs_module_get_config(void) { - return "target_stress (float), copy_matrix (bool), "\ - "ramp_interval (int), tolerance (float), "\ - "max_violations (int), log_interval (int),\n\t"\ - "matrix_size (int)"; + return "pulse_rate (int), high_phase_ratio (float), " + "matrix_size (int), ops_type (string), " + "tolerance (float), sample_interval (int), " + "workload_iterations (int), halt_on_error (bool), " + "gpu_sync_wait (int, reserved), verify_mode (string), " + "max_temp_c (float, 0=off)"; } extern "C" const char* rvs_module_get_output(void) { @@ -77,10 +80,10 @@ extern "C" int rvs_module_terminate(void) { } extern "C" void* rvs_module_action_create(void) { - return static_cast(new perf_action); + return static_cast(new pulse_action); } -extern "C" int rvs_module_action_destroy(void* pAction) { +extern "C" int rvs_module_action_destroy(void* pAction) { delete static_cast(pAction); return 0; } diff --git a/rcqt.so/include/rcutils.h b/rcqt.so/include/rcutils.h index a1ed5a151..a2e3a897c 100644 --- a/rcqt.so/include/rcutils.h +++ b/rcqt.so/include/rcutils.h @@ -46,6 +46,7 @@ enum class OSType { Oracle, Azure, Amazon, + Alibaba, None }; @@ -59,7 +60,8 @@ const std::map op_systems { {"red hat enterprise linux", OSType::RHEL}, {"oracle linux server", OSType::Oracle}, {"microsoft azure linux", OSType::Azure}, - {"amazon linux", OSType::Amazon} + {"amazon linux", OSType::Amazon}, + {"alibaba cloud linux", OSType::Alibaba}, }; struct package_info{ diff --git a/rcqt.so/src/handlerCreator.cpp b/rcqt.so/src/handlerCreator.cpp index b3c0616b1..6ff8ed79e 100644 --- a/rcqt.so/src/handlerCreator.cpp +++ b/rcqt.so/src/handlerCreator.cpp @@ -36,7 +36,7 @@ PackageHandler* handlerCreator::getPackageHandler(const std::string& pkg){ } else if (OSType::Centos == osName || OSType::RHEL == osName || OSType::Oracle == osName || OSType::Azure == osName || - OSType::Amazon == osName) { + OSType::Amazon == osName || OSType::Alibaba == osName) { lptr = new PackageHandlerRpm{pkg}; } else if (OSType::SLES == osName) { @@ -55,7 +55,7 @@ PackageHandler* handlerCreator::getPackageHandler(){ } else if (OSType::Centos == osName || OSType::RHEL == osName || OSType::Oracle == osName || OSType::Azure == osName || - OSType::Amazon == osName) { + OSType::Amazon == osName || OSType::Alibaba == osName) { lptr = new PackageHandlerRpm{}; } else if (OSType::SLES == osName) { diff --git a/rcqt.so/src/rvs_module.cpp b/rcqt.so/src/rvs_module.cpp index 8409e34ea..a0ed909c7 100644 --- a/rcqt.so/src/rvs_module.cpp +++ b/rcqt.so/src/rvs_module.cpp @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2025 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -69,7 +69,6 @@ extern "C" int rvs_module_init(void* pMi) { } extern "C" int rvs_module_terminate(void) { - cleanup_logs(); return 0; } diff --git a/regression/check_json_file.py b/regression/check_json_file.py index 64a7e8354..5dac05560 100755 --- a/regression/check_json_file.py +++ b/regression/check_json_file.py @@ -5,8 +5,8 @@ import sys import json -#print "Number of arguments: ", len(sys.argv) -#print "The arguments are: " , str(sys.argv) +#print("Number of arguments: ", len(sys.argv)) +#print("The arguments are: " , str(sys.argv)) # -------------------- # passed arguments: @@ -23,36 +23,36 @@ if os.path.isfile(json_log_path): f = open(json_log_path) # validate json format - print "check json format" + print("check json format") try: data = json.load(f) json_has_res_err = False for d in data: json_line = d['loglevelname'] - print json_line + print(json_line) if json_line == 'RESULT': - print "JSON Found RESULT" + print("JSON Found RESULT") json_has_res_err = True break if json_line == 'ERROR ': - print "JSON Found ERROR" + print("JSON Found ERROR") json_has_res_err = True break if json_has_res_err == False: - print "JSON No found RESULT/ERROR" + print("JSON No found RESULT/ERROR") test_result = False except ValueError as e: print('Invalid json: %s' % e) test_result = False f.close() else: - print "No file found" + print("No file found") test_result = False # return result if test_result == True: - print json_log_path + " - PASS" + print(json_log_path + " - PASS") sys.exit(0) else: - print json_log_path + " - FAIL" + print(json_log_path + " - FAIL") sys.exit(1) diff --git a/regression/make_ctest_conf.py b/regression/make_ctest_conf.py index b9947a6b1..cf5f22627 100755 --- a/regression/make_ctest_conf.py +++ b/regression/make_ctest_conf.py @@ -26,9 +26,6 @@ ################################################################################### """ -# This script only creates valid combinations, invalid ones will be created as special cases -from __future__ import print_function - import os, fnmatch import itertools import sys diff --git a/regression/make_ctest_conf_logging.py b/regression/make_ctest_conf_logging.py index 8075ce4b8..b4a476245 100755 --- a/regression/make_ctest_conf_logging.py +++ b/regression/make_ctest_conf_logging.py @@ -26,9 +26,6 @@ ################################################################################### """ -# This script only creates valid combinations, invalid ones will be created as special cases -from __future__ import print_function - import os, fnmatch import itertools import sys diff --git a/regression/make_ctest_conf_testif.py b/regression/make_ctest_conf_testif.py index cd626e749..606cdd8bc 100755 --- a/regression/make_ctest_conf_testif.py +++ b/regression/make_ctest_conf_testif.py @@ -26,9 +26,6 @@ ################################################################################### """ -# This script only creates valid combinations, invalid ones will be created as special cases -from __future__ import print_function - import os, fnmatch import itertools import sys diff --git a/regression/make_pbqt_conf.py b/regression/make_pbqt_conf.py index 275930dd9..e1b1c7824 100755 --- a/regression/make_pbqt_conf.py +++ b/regression/make_pbqt_conf.py @@ -1,14 +1,10 @@ #!/usr/bin/env python3 -# This script only creates valid combinations, invalid ones will be created as special cases -from __future__ import print_function - import os import itertools +import random import sys -from random import sample - # global variables module_name = "pbqt" cmake_file_name = "rand_tests.cmake" @@ -19,10 +15,19 @@ for root, dirs, files in os.walk('/sys/class/kfd/kfd/topology/nodes'): for name in files: if name == 'gpu_id': - gpuid = os.popen('cat {}'.format(os.path.join(root, name))).read().rstrip() + with open(os.path.join(root, name)) as gpu_file: + gpuid = gpu_file.read().rstrip() if gpuid not in ['0', '']: - devid = os.popen("grep 'device_id' {} | cut -f 2 -d ' '".format(os.path.join(root, 'properties'))).read().rstrip() - device_id.add(int(devid)) + devid = '' + with open(os.path.join(root, 'properties')) as props_file: + for line in props_file: + if 'device_id' not in line: + continue + # Match: grep 'device_id' | cut -f 2 -d ' ' + fields = line.rstrip('\n').split(' ') + devid = fields[1] if len(fields) > 1 else fields[0] + break + device_id.add(int(devid.rstrip())) gpu_ids.add(int(gpuid)) log_interval = [1000] @@ -129,7 +134,7 @@ f.write(' duration: {}\n'.format(duration_f)) if sample_size: - sample_gpus = sample(gpu_ids, sample_size) + sample_gpus = random.SystemRandom().sample(list(gpu_ids), sample_size) f.write(' peers:') for p in sample_gpus: f.write(' {}'.format(p)) diff --git a/regression/multi_run_and_check_json.py b/regression/multi_run_and_check_json.py index 4d21e4de7..c2e278131 100755 --- a/regression/multi_run_and_check_json.py +++ b/regression/multi_run_and_check_json.py @@ -1,6 +1,6 @@ #!/usr/bin/env python3 -import subprocess +import subprocess # nosec B404 # argument lists only; no shell import os import mmap import sys @@ -8,8 +8,8 @@ # global variables test_output_file_name = "tmp_log_result.txt" -#print "Number of arguments: ", len(sys.argv) -#print "The arguments are: " , str(sys.argv) +#print("Number of arguments: ", len(sys.argv)) +#print("The arguments are: " , str(sys.argv)) # -------------------- # passed arguments: @@ -29,41 +29,39 @@ # check input values if num_runs < 2: - print "number of runs (argument 4) should be at least 2" + print("number of runs (argument 4) should be at least 2") sys.exit(1) if not debug_level in ['0', '1', '2', '3', '4', '5']: - print "debug_level (argument 5) should be inside true /false" + print("debug_level (argument 5) should be inside true /false") sys.exit(1) # ./multi_run_and_check_json.py /work/igorhdl/ROCm2/build/bin /work/igorhdl/ROCm2/ROCmValidationSuite /work/igorhdl/ROCm2/ROCmValidationSuite/rvs/conf/rand_pbqt0.conf 5 3 # get current location -curr_location = subprocess.check_output(["pwd", ""]) -curr_location_size = len(curr_location) -curr_location = curr_location[0:curr_location_size-1] -print curr_location +curr_location = os.getcwd() +print(curr_location) # run test commands result_json = bin_path + "/" + test_output_file_name -test_cmd_init = bin_path + "/rvs -d %s -c %s -l %s -j" % (debug_level, conf_name, result_json) -test_cmd = test_cmd_init + " -a" +rvs_cmd = [os.path.abspath(os.path.join(bin_path, "rvs")), "-d", debug_level, "-c", conf_name, "-l", result_json, "-j"] +rvs_cmd_append = rvs_cmd + ["-a"] # start running tests os.chdir(rvs_path + "/regression") for i in range(0, num_runs): - print "Iteration %d" % (i) + print("Iteration %d" % (i)) if i == 0: - tst_result = os.system(test_cmd_init) + tst_result = subprocess.call(rvs_cmd) # nosec B603 else: - tst_result = os.system(test_cmd) + tst_result = subprocess.call(rvs_cmd_append) # nosec B603 # also check test result - print "Test result is : %s" % (tst_result) - if tst_result > 0: - print "Test is expected to pass with value 0, but return value is %s" %(tst_result) - print conf_name + " - FAIL" + print("Test result is : %s" % (tst_result)) + if tst_result != 0: + print("Test is expected to pass with value 0, but return value is %s" %(tst_result)) + print(conf_name + " - FAIL") sys.exit(1) os.chdir(curr_location) @@ -71,15 +69,16 @@ # result test pass/fail test_result = True -json_result = os.system("./check_json_file.py " + result_json) +json_checker = os.path.join(curr_location, "check_json_file.py") +json_result = subprocess.call([json_checker, result_json]) # nosec B603 if json_result == 1: - print "Json file is invalid" + print("Json file is invalid") test_result = False # return result if test_result == True: - print conf_name + " - PASS" + print(conf_name + " - PASS") sys.exit(0) else: - print conf_name + " - FAIL" + print(conf_name + " - FAIL") sys.exit(1) diff --git a/regression/pbqt_create_conf.py b/regression/pbqt_create_conf.py index 1251fca4c..c1250b5b1 100755 --- a/regression/pbqt_create_conf.py +++ b/regression/pbqt_create_conf.py @@ -3,10 +3,7 @@ # This script only creates valid combinations, invalid ones will be created as special cases import os - -from random import seed -from random import random -from random import sample +import random # global variables module_name = "demofile" @@ -29,7 +26,7 @@ counter = 0 total_iterations = (len(gpu_ids) + 1) * len(log_interval) * len(duration) * len(test_bandwidth) * len(bidirectional) * len(parralel) * len(device_id) -print "Total number of combinations (including invalid) is " + str(total_iterations) +print("Total number of combinations (including invalid) is " + str(total_iterations)) gpu_ids_size = len(gpu_ids) @@ -49,7 +46,7 @@ # for each combination create the conf file filename = conf_location + module_name + str(counter) + ".conf" - print 'Iteration is %d' % (counter) + ", working on conf file " + filename + print('Iteration is %d' % (counter) + ", working on conf file " + filename) f = open(filename, "w") counter = counter + 1 @@ -63,7 +60,7 @@ if sample_size == 0: f.write(" peers: all" + "\n") else: - sample_gpus = sample(gpu_ids, sample_size) + sample_gpus = random.SystemRandom().sample(gpu_ids, sample_size) f.write(" peers:") for p in sample_gpus: f.write(" " + str(p)) diff --git a/regression/run_and_check_test.py b/regression/run_and_check_test.py index dbb086709..9ed700cae 100755 --- a/regression/run_and_check_test.py +++ b/regression/run_and_check_test.py @@ -1,6 +1,6 @@ #!/usr/bin/env python3 -import subprocess +import subprocess # nosec B404 # argument lists only; no shell import os import mmap import sys @@ -9,8 +9,8 @@ test_console_file_name = "tmp_console_result.txt" test_output_file_name = "tmp_log_result.txt" -#print "Number of arguments: ", len(sys.argv) -#print "The arguments are: " , str(sys.argv) +#print("Number of arguments: ", len(sys.argv)) +#print("The arguments are: " , str(sys.argv)) # -------------------- # passed arguments: @@ -31,28 +31,28 @@ console_usage = sys.argv[4] # only true / false log_usage = sys.argv[5] # only true / false json_usage = sys.argv[6] # only true / false -test_pass_fail = sys.argv[7] # only ttp / ttf +expected_result = sys.argv[7] # only ttp / ttf debug_level = sys.argv[8] # only 0,1,2,3,4,5 # check input values if not console_usage in ['true', 'false']: - print "console_usage (argument 4) should be inside true /false" + print("console_usage (argument 4) should be inside true /false") sys.exit(1) if not log_usage in ['true', 'false']: - print "log_usage (argument 5) should be inside true /false" + print("log_usage (argument 5) should be inside true /false") sys.exit(1) if not json_usage in ['true', 'false']: - print "json_usage (argument 6) should be inside true /false" + print("json_usage (argument 6) should be inside true /false") sys.exit(1) -if not test_pass_fail in ['ttp', 'ttf']: - print "test_pass_fail (argument 7) should be inside true /false" +if not expected_result in ['ttp', 'ttf']: + print("expected_result (argument 7) should be inside true /false") sys.exit(1) if not debug_level in ['0', '1', '2', '3', '4', '5']: - print "debug_level (argument 8) should be inside true /false" + print("debug_level (argument 8) should be inside true /false") sys.exit(1) # ./run_and_check_test.py /work/igorhdl/ROCm2/build/bin /work/igorhdl/ROCm2/ROCmValidationSuite /work/igorhdl/ROCm2/ROCmValidationSuite/rvs/conf/rand_pbqt0.conf true true true ttp 3 @@ -61,38 +61,43 @@ # ./run_single_test /work/igorhdl/ROCm2/build/bin /work/igorhdl/ROCm2/ROCmValidationSuite/rvs/conf/rand_pbqt0.conf 3 tmp_output_file.txt true tmp_console_file.txt # get current location -curr_location = subprocess.check_output(["pwd", ""]) -curr_location_size = len(curr_location) -curr_location = curr_location[0:curr_location_size-1] -print curr_location +curr_location = os.getcwd() +print(curr_location) # run test command if log_usage == 'true': - pass_log = bin_path + "/" + test_output_file_name + log_path = bin_path + "/" + test_output_file_name else: - pass_log = "no_log" - -test_cmd = "./run_single_test %s %s %s %s %s %s" % (bin_path, conf_name, debug_level, pass_log, json_usage, bin_path + "/" + test_console_file_name) + log_path = "no_log" os.chdir(rvs_path + "/regression") -tst_result = os.system(test_cmd) -print "Test result is : %s" % (tst_result) +run_single_test = os.path.join(rvs_path, "regression", "run_single_test") +tst_result = subprocess.call([ # nosec B603 + run_single_test, + bin_path, + conf_name, + debug_level, + log_path, + json_usage, + bin_path + "/" + test_console_file_name, +]) +print("Test result is : %s" % (tst_result)) os.chdir(curr_location) # check test to pass/fail first -if test_pass_fail == 'ttp' and tst_result > 0: - print "Test is expected to pass with value 0, but return value is %s" %(tst_result) - print conf_name + " - FAIL" +if expected_result == 'ttp' and tst_result != 0: + print("Test is expected to pass with value 0, but return value is %s" %(tst_result)) + print(conf_name + " - FAIL") sys.exit(1) -if test_pass_fail == 'ttf': +if expected_result == 'ttf': if tst_result == 0: - print "Test is expected to fail with value different than 0, but return value is %s" %(tst_result) - print conf_name + " - FAIL" + print("Test is expected to fail with value different than 0, but return value is %s" %(tst_result)) + print(conf_name + " - FAIL") sys.exit(1) else: - print "Test is expected to fail and return value is non 0" - print conf_name + " - PASS" + print("Test is expected to fail and return value is non 0") + print(conf_name + " - PASS") sys.exit(0) # result test pass/fail @@ -100,57 +105,58 @@ # check console output if console_usage == 'true': - print "console_usage is True" + print("console_usage is True") result_log = bin_path + "/" + test_console_file_name if os.path.isfile(result_log): if os.path.getsize(result_log) > 0: f = open(result_log) s = mmap.mmap(f.fileno(), 0, access=mmap.ACCESS_READ) - if s.find('RESULT') == -1 and s.find('ERROR') == -1: - print "No found RESULT/ERROR" + if s.find(b'RESULT') == -1 and s.find(b'ERROR') == -1: + print("No found RESULT/ERROR") test_result = False f.close() else: - print "Empty file" + print("Empty file") test_result = False else: - print "No file found" + print("No file found") test_result = False # check json output file if json_usage == 'true' and log_usage == 'true': - print "json_usage is True and log_usage is True" + print("json_usage is True and log_usage is True") result_json = bin_path + "/" + test_output_file_name - json_result = os.system("./check_json_file.py " + result_json) + json_checker = os.path.join(curr_location, "check_json_file.py") + json_result = subprocess.call([json_checker, result_json]) # nosec B603 if json_result == 1: - print "Json file is invalid" + print("Json file is invalid") test_result = False # check console output file else: if log_usage == 'true': - print "log_usage is True" + print("log_usage is True") result_log = bin_path + "/" + test_output_file_name if os.path.isfile(result_log): if os.path.getsize(result_log) > 0: f = open(result_log) s = mmap.mmap(f.fileno(), 0, access=mmap.ACCESS_READ) if s.find('RESULT') == -1 and s.find('ERROR') == -1: - print "No found RESULT/ERROR" + print("No found RESULT/ERROR") test_result = False f.close() else: - print "Empty file" + print("Empty file") test_result = False else: - print "No file found" + print("No file found") test_result = False # return result if test_result == True: - print conf_name + " - PASS" + print(conf_name + " - PASS") sys.exit(0) else: - print conf_name + " - FAIL" + print(conf_name + " - FAIL") sys.exit(1) diff --git a/regression/run_regression.py b/regression/run_regression.py index 2e3a1de6f..a6fecf799 100755 --- a/regression/run_regression.py +++ b/regression/run_regression.py @@ -1,15 +1,13 @@ #!/usr/bin/env python3 -import subprocess +import subprocess # nosec B404 # argument lists only; no shell import os import mmap import sys from shutil import copyfile -curr_location = subprocess.check_output(["pwd", ""]) -curr_location_size = len(curr_location) -curr_location = curr_location[0:curr_location_size-1] +curr_location = os.getcwd() print('curr_location',curr_location) # set paths to build and ROCmValidationSuite folders @@ -39,9 +37,9 @@ print('conf_files',conf_files) # make them executable -os.chdir(build_location) -subprocess.call(["chmod", "+x", "build"]) -subprocess.call(["chmod", "+x", "run"]) +for script_name in ("build", "run"): + script_path = os.path.join(build_location, script_name) + os.chmod(script_path, os.stat(script_path).st_mode | 0o111) if not os.path.exists(regression_directory): os.makedirs(regression_directory) @@ -62,7 +60,11 @@ print('build_location',build_location) # run test os.chdir(build_location) - os.system("./run %s %s" % (conf_location + confname , regression_directory + "/log_" + confname + ".txt")) + subprocess.call([ # nosec B603 + os.path.join(build_location, "run"), + conf_location + confname, + regression_directory + "/log_" + confname + ".txt", + ]) # check json output result_json = regression_directory + "/log_" + confname + ".txt" diff --git a/rvs/.rvsmodules.config b/rvs/.rvsmodules.config index ba753ec99..2c7bacb30 100644 --- a/rvs/.rvsmodules.config +++ b/rvs/.rvsmodules.config @@ -1,15 +1,12 @@ version: 1 gpup: libgpup.so peqt: libpeqt.so -pesm: libpesm.so rcqt: librcqt.so -smqt: libsmqt.so -gm: libgm.so gst: libgst.so pbqt: libpbqt.so pebb: libpebb.so iet: libiet.so mem: libmem.so babel: libbabel.so -perf: libperf.so tst: libtst.so +pulse: libpulse.so diff --git a/rvs/CMakeLists.txt b/rvs/CMakeLists.txt index 03c2a0678..986fb0389 100644 --- a/rvs/CMakeLists.txt +++ b/rvs/CMakeLists.txt @@ -43,7 +43,7 @@ add_compile_options(-pthread) add_compile_options(-Wall) add_compile_options(-DRVS_OS_TYPE_NUM=${RVS_OS_TYPE_NUM}) -find_package(OpenMP) + if (RVS_COVERAGE) add_compile_options(-o0 -fprofile-arcs -ftest-coverage) @@ -124,11 +124,11 @@ set(CORE_RUNTIME_NAME "hsa-runtime") set(HIPRAND_LIB "hiprand") set(HIPBLASLT_LIB "hipblaslt") set(CORE_RUNTIME_TARGET "${CORE_RUNTIME_NAME}64") -set(PROJECT_LINK_LIBS libdl.so libpthread.so libpci.so ${YAML_CPP_LIBRARIES}) +set(PROJECT_LINK_LIBS libdl.so libpthread.so ${LIBPCI_TARGET} ${YAML_CPP_TARGET}) ## define target add_executable(${RVS_TARGET} src/rvs.cpp) -target_link_libraries(${RVS_TARGET} rvslib OpenMP::OpenMP_CXX +target_link_libraries(${RVS_TARGET} rvslib -fopenmp ${ROCBLAS_LIB} ${AMD_SMI_LIB} ${CORE_RUNTIME_TARGET} ${ROCM_CORE} ${PROJECT_LINK_LIBS} ${HIPRAND_LIB} ${HIPBLASLT_LIB}) add_dependencies(${RVS_TARGET} rvslib) @@ -139,13 +139,13 @@ install(TARGETS ${RVS_TARGET} ) install(DIRECTORY ${CMAKE_CURRENT_SOURCE_DIR}/conf - DESTINATION ${CPACK_PACKAGING_INSTALL_PREFIX}/${CMAKE_INSTALL_DATADIR}/${CPACK_PACKAGE_NAME}/ + DESTINATION ${CPACK_PACKAGING_INSTALL_PREFIX}/${CMAKE_INSTALL_DATADIR}/${CMAKE_PROJECT_NAME}/ COMPONENT applications FILES_MATCHING PATTERN "*.conf" ) install(FILES "${CMAKE_CURRENT_SOURCE_DIR}/.rvsmodules.config" - DESTINATION ${CPACK_PACKAGING_INSTALL_PREFIX}/${CMAKE_INSTALL_DATADIR}/${CPACK_PACKAGE_NAME}/conf + DESTINATION ${CPACK_PACKAGING_INSTALL_PREFIX}/${CMAKE_INSTALL_DATADIR}/${CMAKE_PROJECT_NAME}/conf COMPONENT applications ) diff --git a/rvs/conf/MI210/babel.conf b/rvs/conf/MI210/babel.conf index 6602ad933..69c1c02b7 100644 --- a/rvs/conf/MI210/babel.conf +++ b/rvs/conf/MI210/babel.conf @@ -1,6 +1,6 @@ # ################################################################################ # # -# # Copyright (c) 2024 Advanced Micro Devices, Inc. All rights reserved. +# # Copyright (c) 2024-2026 Advanced Micro Devices, Inc. All rights reserved. # # # # MIT LICENSE: # # Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -41,11 +41,16 @@ actions: parallel: true # Parallel true or false count: 1 # Number of times you want to repeat the test from the begin ( A clean start every time) num_iter: 5000 # Number of iterations, this many kernels are launched simultaneosuly and stresses the system + duration: 0 # Duration in milliseconds. When > 0, runs for this time instead of num_iter. 0 = use num_iter array_size: 268435456 # Buffer size the test operates, this is 256 MiB test_type: 1 # type of test, 1: Float, 2: Double, 3: Triad float, 4: Triad double mibibytes: true # mibibytes (MiB) or megabytes (MB), true for MiB o/p_csv: false # o/p as csv file - subtest: 5 # 1: copy 2: copy+mul 3: copy+mul+add 4: copy+mul+add+traid 5: copy+mul+add+traid+dot + copy: true # copy operation + mul: true # mul operation + add: true # add operation + triad: true # triad operation + dot: true # dot operation dwords_per_lane: 4 # Number of dwords per lane chunks_per_block: 4 # Number of chunks per block diff --git a/rvs/conf/smqt_single.conf b/rvs/conf/MI250X/iet_stress.conf similarity index 67% rename from rvs/conf/smqt_single.conf rename to rvs/conf/MI250X/iet_stress.conf index 7bbd8948b..252a8fc0f 100644 --- a/rvs/conf/smqt_single.conf +++ b/rvs/conf/MI250X/iet_stress.conf @@ -1,6 +1,6 @@ # ################################################################################ # # -# # Copyright (c) 2018-2023 Advanced Micro Devices, Inc. All rights reserved. +# # Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. # # # # MIT LICENSE: # # Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -23,36 +23,42 @@ # # # ############################################################################### -# SMQT test +# IET stress test # # Preconditions: # Set device to all. If you need to run the rvs only on a subset of GPUs, please run rvs with -g # option, collect the GPUs IDs (e.g.: GPU[ 5 - 50599] -> 50599 is the GPU ID) and then specify -# all the GPUs IDs separated by white space. -# Set each BAR mapping requirements. +# all the GPUs IDs separated by comma. +# Set parallel execution to true (gemm workload execution on all GPUs in parallel) +# Set gemm operation type as dgemm. +# Set matrix_size to 28000. +# Test duration set to 10 mins. +# Target power set to 560W for each GPU. # # Run test with: # cd bin -# sudo ./rvs -c conf/smqt_single.conf -d 3 +# ./rvs -c conf/MI250X/iet_stress.conf # # Expected result: -# True if platform’s SBIOS has satisfied the BAR mapping requirements for each GPUs. +# The test on each GPU passes (TRUE) if the GPU achieves power target of 560W. # -# Note: BAR requirements are platform specific so peoper values has be set based on it. -# Below values are just sample arbitrary values. actions: -- name: bar_qualification +- name: iet-stress-560W-dgemm-true device: all - module: smqt - bar1_req_size: 17179869184 - bar1_base_addr_min: 0 - bar1_base_addr_max: 17592168044416 - bar2_req_size: 2097152 - bar2_base_addr_min: 0 - bar2_base_addr_max: 1099511627776 - bar4_req_size: 262144 - bar4_base_addr_min: 0 - bar4_base_addr_max: 17592168044416 - bar5_req_size: 131072 + module: iet + parallel: true + duration: 600000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + target_power: 560 + matrix_size: 28000 + ops_type: dgemm + lda: 28000 + ldb: 28000 + ldc: 28000 + alpha: 1 + beta: 1 + matrix_init: hiprand diff --git a/rvs/conf/MI300X-HF/iet_stress.conf b/rvs/conf/MI300X-HF/iet_stress.conf index 952c3c0de..5cc9ed6e4 100644 --- a/rvs/conf/MI300X-HF/iet_stress.conf +++ b/rvs/conf/MI300X-HF/iet_stress.conf @@ -1,6 +1,6 @@ # ################################################################################ # # -# # Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. +# # Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. # # # # MIT LICENSE: # # Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -34,6 +34,7 @@ # Set matrix_size to 28000. # Test duration set to 10 mins. # Target power set to 850W for each GPU. +# Power tolerance of 1% (8.5W). # # Run test with: # cd bin @@ -53,6 +54,7 @@ actions: sample_interval: 5000 log_interval: 5000 target_power: 850 + tolerance: 0.01 matrix_size: 28000 ops_type: dgemm lda: 28000 diff --git a/rvs/conf/MI300X-HF/levels/rvs_level_1.conf b/rvs/conf/MI300X-HF/levels/rvs_level_1.conf new file mode 100644 index 000000000..0d3512230 --- /dev/null +++ b/rvs/conf/MI300X-HF/levels/rvs_level_1.conf @@ -0,0 +1,38 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: + +- name: metapackage-validation + device: all + module: rcqt + package: rocm-hip-sdk rocm-hip-libraries rocm-ml-libraries rocm-ml-sdk amdgpu-dkms rocm-language-runtime rocm-openmp-sdk rocm-utils rocm-opencl-sdk rocm-opencl-runtime rocm-hip-runtime rocm-developer-tools + +- name: packagelist-install-validation + device: all + module: rcqt + rpmpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-devel rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-devel hipcub-devel rocsolver rocsparse rocsolver-devel rocminfo hipfft-devel rocm-gdb rocm-dbgapi rocfft hipblas-devel rocthrust-devel openmp-extras-runtime openmp-extras-devel comgr rccl rocblas hipblas roctracer-devel hip-doc rocrand hsa-rocr hipfft hipsparse-devel rocsparse-devel rocrand-devel rocm-opencl hip-devel rocprim-devel hipsolver-devel rocfft-devel hsa-amd-aqlprofile hipify-clang miopen-hip-devel rocm-llvm hip-runtime-amd hip-samples rocalution-devel rccl-devel hipsolver rocprofiler-devel miopen-hip rocm-cmake hipsparse rocblas-devel rocm-opencl-devel + debpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-dev rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-dev hipcub-dev rocsolver rocsparse rocsolver-dev rocminfo hipfft-dev rocm-gdb rocm-dbgapi rocfft hipblas-dev rocthrust-dev openmp-extras-runtime openmp-extras-dev comgr rccl rocblas hipblas roctracer-dev hip-doc rocrand hsa-rocr hipfft hipsparse-dev rocsparse-dev rocrand-dev rocm-opencl hip-dev rocprim-dev hipsolver-dev rocfft-dev hsa-amd-aqlprofile hipify-clang miopen-hip-dev rocm-llvm hip-runtime-amd hip-samples rocalution-dev rccl-dev hipsolver rocprofiler-dev miopen-hip rocm-cmake hipsparse rocblas-dev rocm-opencl-dev + diff --git a/rvs/conf/MI300X-HF/levels/rvs_level_2.conf b/rvs/conf/MI300X-HF/levels/rvs_level_2.conf new file mode 100644 index 000000000..fb3e444cb --- /dev/null +++ b/rvs/conf/MI300X-HF/levels/rvs_level_2.conf @@ -0,0 +1,75 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: pcie_capability + device: all + module: peqt + capability: + link_cap_max_speed: + link_cap_max_width: + link_stat_cur_speed: + link_stat_neg_width: + slot_pwr_limit_value: + slot_physical_num: + deviceid: + vendor_id: + kernel_driver: + +- name: hbm_basic + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 268435456 + test_type: 1 + mibibytes: true + o/p_csv: false + read: true + write: true + +- name: xgmi_interface + device: all + module: pbqt + log_interval: 5000 + duration: 20000 + peers: all + test_bandwidth: true + bidirectional: false + parallel: true + block_size: 67108864 + device_id: all + +- name: pcie_interface + device: all + module: pebb + duration: 20000 + device_to_host: false + host_to_device: true + parallel: true + block_size: 67108864 + link_type: 2 + diff --git a/rvs/conf/MI300X-HF/levels/rvs_level_3.conf b/rvs/conf/MI300X-HF/levels/rvs_level_3.conf new file mode 100644 index 000000000..ddb35de8e --- /dev/null +++ b/rvs/conf/MI300X-HF/levels/rvs_level_3.conf @@ -0,0 +1,150 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: hbm_mid + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 268435456 + test_type: 1 + mibibytes: true + o/p_csv: false + read: true + write: true + copy: true + add: true + mul: true + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 981000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp8_r + out_data_type: fp8_r + compute_type: fp32_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 523000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 552000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: bf16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + diff --git a/rvs/conf/MI300X-HF/levels/rvs_level_4.conf b/rvs/conf/MI300X-HF/levels/rvs_level_4.conf new file mode 100644 index 000000000..70a28ac4b --- /dev/null +++ b/rvs/conf/MI300X-HF/levels/rvs_level_4.conf @@ -0,0 +1,229 @@ +# ################################################################################ +# # +# # Copyright (c) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: hbm_full + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 268435456 + test_type: 1 + mibibytes: true + o/p_csv: false + read: true + write: true + copy: true + add: true + mul: true + triad: true + dot: true + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: memtest + device: all + module: mem + parallel: true + count: 1 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 981000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp8_r + out_data_type: fp8_r + compute_type: fp32_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 523000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 552000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: bf16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-fp32-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 100000 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: trig + ops_type: sgemm + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-fp64-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 70000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: trig + ops_type: dgemm + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: power-stress + device: all + module: iet + parallel: true + duration: 60000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + target_power: 850 + tolerance: 0.01 + matrix_size: 28000 + ops_type: dgemm + lda: 28000 + ldb: 28000 + ldc: 28000 + alpha: 1 + beta: 1 + matrix_init: hiprand + diff --git a/rvs/conf/MI300X-HF/levels/rvs_level_5.conf b/rvs/conf/MI300X-HF/levels/rvs_level_5.conf new file mode 100644 index 000000000..2e0ceaeef --- /dev/null +++ b/rvs/conf/MI300X-HF/levels/rvs_level_5.conf @@ -0,0 +1,229 @@ +# ################################################################################ +# # +# # Copyright (c) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: hbm_full + device: all + module: babel + parallel: true + count: 1 + num_iter: 10000 + array_size: 268435456 + test_type: 1 + mibibytes: true + o/p_csv: false + read: true + write: true + copy: true + add: true + mul: true + triad: true + dot: true + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 900000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: memtest + device: all + module: mem + parallel: true + count: 20 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 981000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp8_r + out_data_type: fp8_r + compute_type: fp32_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 523000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 552000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: bf16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-fp32-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 100000 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: trig + ops_type: sgemm + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-fp64-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 70000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: trig + ops_type: dgemm + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: power-stress + device: all + module: iet + parallel: true + duration: 3600000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + target_power: 850 + tolerance: 0.01 + matrix_size: 28000 + ops_type: dgemm + lda: 28000 + ldb: 28000 + ldc: 28000 + alpha: 1 + beta: 1 + matrix_init: hiprand + diff --git a/rvs/conf/MI300X/NPS2/DPX/gst_single.conf b/rvs/conf/MI300X/NPS2/DPX/gst_single.conf index 1f9cfa9f7..b525c5d7c 100644 --- a/rvs/conf/MI300X/NPS2/DPX/gst_single.conf +++ b/rvs/conf/MI300X/NPS2/DPX/gst_single.conf @@ -1,6 +1,6 @@ # ################################################################################ # # -# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # Copyright (c) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. # # # # MIT LICENSE: # # Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -23,7 +23,7 @@ # # # ############################################################################### -# GST test - gst-719Tflops-4K4K8K-rand-fp8 +# GST test - gst-Tflops-4K4K8K-rand-fp8 # # Preconditions: # Set device to all. If you need to run the rvs only on a subset of GPUs, please run rvs with -g @@ -33,15 +33,13 @@ # Set matrix data type as fp8 real number # Set matrix data initialization method as random integer # Set copy_matrix to false (the matrices will be copied to GPUs only once) -# Set target stress GFLOPS as 719000 (719 TFLOPS) # # Expected result: -# The test on each GPU passes (TRUE) if the GPU achieves 719 TFLOPS or more -# within the test duration of 15 seconds after ramp-up duration of 5 seconds. +# The test on each GPU passes (TRUE) if the GPU completes the operation successfully. # Else test on the GPU fails (FALSE). actions: -- name: gst-719Tflops-4K4K8K-rand-fp8 +- name: gst-Tflops-4K4K8K-rand-fp8 device: all module: gst log_interval: 2000 @@ -49,12 +47,14 @@ actions: duration: 5000 hot_calls: 1000 copy_matrix: false - target_stress: 719000 + target_stress: 0 matrix_size_a: 4864 matrix_size_b: 4096 matrix_size_c: 8192 matrix_init: rand data_type: fp8_r + out_data_type: fp8_r + compute_type: fp32_r lda: 8320 ldb: 8320 ldc: 4992 @@ -63,8 +63,9 @@ actions: transb: 0 alpha: 1 beta: 0 + blas_source: hipblaslt -- name: gst-491Tflops-4K4K8K-trig-fp8 +- name: gst-Tflops-4K4K8K-trig-fp8 device: all module: gst log_interval: 2000 @@ -72,12 +73,14 @@ actions: duration: 5000 hot_calls: 1000 copy_matrix: false - target_stress: 491000 + target_stress: 0 matrix_size_a: 4864 matrix_size_b: 4096 matrix_size_c: 8192 matrix_init: trig data_type: fp8_r + out_data_type: fp8_r + compute_type: fp32_r lda: 8320 ldb: 8320 ldc: 4992 @@ -86,8 +89,9 @@ actions: transb: 0 alpha: 1 beta: 0 + blas_source: hipblaslt -- name: gst-348Tflops-4K4K8K-rand-fp16 +- name: gst-Tflops-4K4K8K-rand-fp16 device: all module: gst log_interval: 2000 @@ -95,7 +99,7 @@ actions: duration: 5000 hot_calls: 1000 copy_matrix: false - target_stress: 348000 + target_stress: 0 matrix_size_a: 4864 matrix_size_b: 4096 matrix_size_c: 8192 @@ -110,7 +114,7 @@ actions: alpha: 1 beta: 0 -- name: gst-261Tflops-4K4K8K-trig-fp16 +- name: gst-Tflops-4K4K8K-trig-fp16 device: all module: gst log_interval: 2000 @@ -118,7 +122,7 @@ actions: duration: 5000 hot_calls: 1000 copy_matrix: false - target_stress: 261500 + target_stress: 0 matrix_size_a: 4864 matrix_size_b: 4096 matrix_size_c: 8192 @@ -133,7 +137,7 @@ actions: alpha: 1 beta: 0 -- name: gst-334Tflops-4K4K8K-rand-bf16 +- name: gst-Tflops-4K4K8K-rand-bf16 device: all module: gst log_interval: 2000 @@ -141,7 +145,7 @@ actions: duration: 5000 hot_calls: 1000 copy_matrix: false - target_stress: 334500 + target_stress: 0 matrix_size_a: 4864 matrix_size_b: 4096 matrix_size_c: 8192 @@ -156,7 +160,7 @@ actions: alpha: 1 beta: 0 -- name: gst-276Tflops-4K4K8K-trig-bf16 +- name: gst-Tflops-4K4K8K-trig-bf16 device: all module: gst log_interval: 2000 @@ -164,7 +168,7 @@ actions: duration: 5000 hot_calls: 1000 copy_matrix: false - target_stress: 276500 + target_stress: 0 matrix_size_a: 4864 matrix_size_b: 4096 matrix_size_c: 8192 @@ -179,7 +183,7 @@ actions: alpha: 1 beta: 0 -- name: gst-53Tflops-3K-trig-sgemm +- name: gst-Tflops-3K-trig-sgemm device: all module: gst log_interval: 2000 @@ -187,7 +191,7 @@ actions: duration: 5000 hot_calls: 1000 copy_matrix: false - target_stress: 53500 + target_stress: 0 matrix_size_a: 3072 matrix_size_b: 3072 matrix_size_c: 3072 @@ -201,7 +205,7 @@ actions: alpha: 1 beta: 0 -- name: gst-53Tflops-3K-rand-sgemm +- name: gst-Tflops-3K-rand-sgemm device: all module: gst log_interval: 2000 @@ -209,7 +213,7 @@ actions: duration: 5000 hot_calls: 1000 copy_matrix: false - target_stress: 53500 + target_stress: 0 matrix_size_a: 3072 matrix_size_b: 3072 matrix_size_c: 3072 @@ -223,7 +227,7 @@ actions: alpha: 1 beta: 0 -- name: gst-35Tflops-8K-trig-dgemm +- name: gst-Tflops-8K-trig-dgemm device: all module: gst log_interval: 2000 @@ -231,7 +235,7 @@ actions: duration: 5000 hot_calls: 1000 copy_matrix: false - target_stress: 35500 + target_stress: 0 matrix_size_a: 8192 matrix_size_b: 8192 matrix_size_c: 8192 @@ -245,7 +249,7 @@ actions: alpha: 1 beta: 0 -- name: gst-35Tflops-8K-rand-dgemm +- name: gst-Tflops-8K-rand-dgemm device: all module: gst log_interval: 2000 @@ -253,7 +257,7 @@ actions: duration: 5000 hot_calls: 1000 copy_matrix: false - target_stress: 35500 + target_stress: 0 matrix_size_a: 8192 matrix_size_b: 8192 matrix_size_c: 8192 diff --git a/rvs/conf/MI300X/NPS4/CPX/gst_single.conf b/rvs/conf/MI300X/NPS4/CPX/gst_single.conf index b59ba7fc1..c8d3de9d2 100644 --- a/rvs/conf/MI300X/NPS4/CPX/gst_single.conf +++ b/rvs/conf/MI300X/NPS4/CPX/gst_single.conf @@ -1,6 +1,6 @@ # ################################################################################ # # -# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # Copyright (c) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. # # # # MIT LICENSE: # # Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -23,7 +23,7 @@ # # # ############################################################################### -# GST test - gst-179Tflops-4K4K8K-rand-fp8 +# GST test - gst-Tflops-4K4K8K-rand-fp8 # # Preconditions: # Set device to all. If you need to run the rvs only on a subset of GPUs, please run rvs with -g @@ -33,15 +33,13 @@ # Set matrix data type as fp8 real number # Set matrix data initialization method as random integer # Set copy_matrix to false (the matrices will be copied to GPUs only once) -# Set target stress GFLOPS as 179750 (179 TFLOPS) # # Expected result: -# The test on each GPU passes (TRUE) if the GPU achieves 179 TFLOPS or more -# within the test duration of 15 seconds after ramp-up duration of 5 seconds. +# The test on each GPU passes (TRUE) if the GPU completes the operation successfully. # Else test on the GPU fails (FALSE). actions: -- name: gst-179Tflops-4K4K8K-rand-fp8 +- name: gst-Tflops-4K4K8K-rand-fp8 device: all module: gst log_interval: 1000 @@ -49,12 +47,14 @@ actions: duration: 1000 hot_calls: 100 copy_matrix: false - target_stress: 179750 + target_stress: 0 matrix_size_a: 4864 matrix_size_b: 4096 matrix_size_c: 8192 matrix_init: rand data_type: fp8_r + out_data_type: fp8_r + compute_type: fp32_r lda: 8320 ldb: 8320 ldc: 4992 @@ -63,8 +63,9 @@ actions: transb: 0 alpha: 1 beta: 0 + blas_source: hipblaslt -- name: gst-122Tflops-4K4K8K-trig-fp8 +- name: gst-Tflops-4K4K8K-trig-fp8 device: all module: gst log_interval: 1000 @@ -72,12 +73,14 @@ actions: duration: 1000 hot_calls: 100 copy_matrix: false - target_stress: 122750 + target_stress: 0 matrix_size_a: 4864 matrix_size_b: 4096 matrix_size_c: 8192 matrix_init: trig data_type: fp8_r + out_data_type: fp8_r + compute_type: fp32_r lda: 8320 ldb: 8320 ldc: 4992 @@ -86,8 +89,9 @@ actions: transb: 0 alpha: 1 beta: 0 + blas_source: hipblaslt -- name: gst-87Tflops-4K4K8K-rand-fp16 +- name: gst-Tflops-4K4K8K-rand-fp16 device: all module: gst log_interval: 1000 @@ -95,7 +99,7 @@ actions: duration: 1000 hot_calls: 100 copy_matrix: false - target_stress: 87000 + target_stress: 0 matrix_size_a: 4864 matrix_size_b: 4096 matrix_size_c: 8192 @@ -110,7 +114,7 @@ actions: alpha: 1 beta: 0 -- name: gst-65Tflops-4K4K8K-trig-fp16 +- name: gst-Tflops-4K4K8K-trig-fp16 device: all module: gst log_interval: 1000 @@ -118,7 +122,7 @@ actions: duration: 1000 hot_calls: 100 copy_matrix: false - target_stress: 65375 + target_stress: 0 matrix_size_a: 4864 matrix_size_b: 4096 matrix_size_c: 8192 @@ -133,7 +137,7 @@ actions: alpha: 1 beta: 0 -- name: gst-83Tflops-4K4K8K-rand-bf16 +- name: gst-Tflops-4K4K8K-rand-bf16 device: all module: gst log_interval: 1000 @@ -141,7 +145,7 @@ actions: duration: 1000 hot_calls: 100 copy_matrix: false - target_stress: 83625 + target_stress: 0 matrix_size_a: 4864 matrix_size_b: 4096 matrix_size_c: 8192 @@ -156,14 +160,15 @@ actions: alpha: 1 beta: 0 -- name: gst-69Tflops-4K4K8K-trig-bf16 +- name: gst-Tflops-4K4K8K-trig-bf16 device: all module: gst + log_interval: 1000 ramp_interval: 500 duration: 1000 hot_calls: 100 copy_matrix: false - target_stress: 69125 + target_stress: 0 matrix_size_a: 4864 matrix_size_b: 4096 matrix_size_c: 8192 @@ -178,7 +183,7 @@ actions: alpha: 1 beta: 0 -- name: gst-13Tflops-3K-trig-sgemm +- name: gst-Tflops-3K-trig-sgemm device: all module: gst log_interval: 1000 @@ -186,7 +191,7 @@ actions: duration: 1000 hot_calls: 100 copy_matrix: false - target_stress: 13375 + target_stress: 0 matrix_size_a: 3072 matrix_size_b: 3072 matrix_size_c: 3072 @@ -200,16 +205,15 @@ actions: alpha: 1 beta: 0 -- name: gst-13Tflops-3K-rand-sgemm +- name: gst-Tflops-3K-rand-sgemm device: all module: gst - hot_calls: 1000 log_interval: 1000 ramp_interval: 500 duration: 1000 hot_calls: 100 copy_matrix: false - target_stress: 13375 + target_stress: 0 matrix_size_a: 3072 matrix_size_b: 3072 matrix_size_c: 3072 @@ -223,7 +227,7 @@ actions: alpha: 1 beta: 0 -- name: gst-8Tflops-8K-trig-dgemm +- name: gst-Tflops-8K-trig-dgemm device: all module: gst log_interval: 1000 @@ -231,7 +235,7 @@ actions: duration: 1000 hot_calls: 100 copy_matrix: false - target_stress: 8875 + target_stress: 0 matrix_size_a: 8192 matrix_size_b: 8192 matrix_size_c: 8192 @@ -245,7 +249,7 @@ actions: alpha: 1 beta: 0 -- name: gst-8Tflops-8K-rand-dgemm +- name: gst-Tflops-8K-rand-dgemm device: all module: gst log_interval: 1000 @@ -253,7 +257,7 @@ actions: duration: 1000 hot_calls: 100 copy_matrix: false - target_stress: 8875 + target_stress: 0 matrix_size_a: 8192 matrix_size_b: 8192 matrix_size_c: 8192 diff --git a/rvs/conf/MI300X/babel.conf b/rvs/conf/MI300X/babel.conf index 22d2c6a4b..2c06833a8 100644 --- a/rvs/conf/MI300X/babel.conf +++ b/rvs/conf/MI300X/babel.conf @@ -1,6 +1,6 @@ # ################################################################################ # # -# # Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. +# # Copyright (c) 2024-2026 Advanced Micro Devices, Inc. All rights reserved. # # # # MIT LICENSE: # # Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -41,9 +41,14 @@ actions: parallel: false # Parallel true or false count: 1 # Number of times you want to repeat the test from the begin ( A clean start every time) num_iter: 5000 # Number of iterations, this many kernels are launched simultaneosuly and stresses the system + duration: 0 # Duration in milliseconds. When > 0, runs for this time instead of num_iter. 0 = use num_iter array_size: 268435456 # Buffer size the test operates, this is 256 MiB test_type: 1 # type of test, 1: Float, 2: Double, 3: Triad float, 4: Triad double mibibytes: true # mibibytes (MiB) or megabytes (MB), true for MiB o/p_csv: false # o/p as csv file - subtest: 5 # 1: copy 2: copy+mul 3: copy+mul+add 4: copy+mul+add+traid 5: copy+mul+add+traid+dot + copy: true # copy operation + mul: true # mul operation + add: true # add operation + dot: true # dot operation + triad: true # triad operation diff --git a/rvs/conf/MI300X/gst_selfcheck.conf b/rvs/conf/MI300X/gst_selfcheck.conf index a7a417adf..1376af879 100644 --- a/rvs/conf/MI300X/gst_selfcheck.conf +++ b/rvs/conf/MI300X/gst_selfcheck.conf @@ -178,4 +178,5 @@ actions: error_inject: true error_freq: 2 error_count: 1 + blas_source: hipblaslt diff --git a/rvs/conf/MI300X/iet_stress.conf b/rvs/conf/MI300X/iet_stress.conf index fa2885e3e..8545e3080 100644 --- a/rvs/conf/MI300X/iet_stress.conf +++ b/rvs/conf/MI300X/iet_stress.conf @@ -1,6 +1,6 @@ # ################################################################################ # # -# # Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. +# # Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. # # # # MIT LICENSE: # # Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -34,6 +34,7 @@ # Set matrix_size to 28000. # Test duration set to 10 mins. # Target power set to 750W for each GPU. +# Power tolerance of 1% (7.5W). # # Run test with: # cd bin @@ -53,6 +54,7 @@ actions: sample_interval: 5000 log_interval: 5000 target_power: 750 + tolerance: 0.01 matrix_size: 28000 ops_type: dgemm lda: 28000 diff --git a/rvs/conf/MI300X/levels/rvs_level_1.conf b/rvs/conf/MI300X/levels/rvs_level_1.conf new file mode 100644 index 000000000..0d3512230 --- /dev/null +++ b/rvs/conf/MI300X/levels/rvs_level_1.conf @@ -0,0 +1,38 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: + +- name: metapackage-validation + device: all + module: rcqt + package: rocm-hip-sdk rocm-hip-libraries rocm-ml-libraries rocm-ml-sdk amdgpu-dkms rocm-language-runtime rocm-openmp-sdk rocm-utils rocm-opencl-sdk rocm-opencl-runtime rocm-hip-runtime rocm-developer-tools + +- name: packagelist-install-validation + device: all + module: rcqt + rpmpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-devel rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-devel hipcub-devel rocsolver rocsparse rocsolver-devel rocminfo hipfft-devel rocm-gdb rocm-dbgapi rocfft hipblas-devel rocthrust-devel openmp-extras-runtime openmp-extras-devel comgr rccl rocblas hipblas roctracer-devel hip-doc rocrand hsa-rocr hipfft hipsparse-devel rocsparse-devel rocrand-devel rocm-opencl hip-devel rocprim-devel hipsolver-devel rocfft-devel hsa-amd-aqlprofile hipify-clang miopen-hip-devel rocm-llvm hip-runtime-amd hip-samples rocalution-devel rccl-devel hipsolver rocprofiler-devel miopen-hip rocm-cmake hipsparse rocblas-devel rocm-opencl-devel + debpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-dev rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-dev hipcub-dev rocsolver rocsparse rocsolver-dev rocminfo hipfft-dev rocm-gdb rocm-dbgapi rocfft hipblas-dev rocthrust-dev openmp-extras-runtime openmp-extras-dev comgr rccl rocblas hipblas roctracer-dev hip-doc rocrand hsa-rocr hipfft hipsparse-dev rocsparse-dev rocrand-dev rocm-opencl hip-dev rocprim-dev hipsolver-dev rocfft-dev hsa-amd-aqlprofile hipify-clang miopen-hip-dev rocm-llvm hip-runtime-amd hip-samples rocalution-dev rccl-dev hipsolver rocprofiler-dev miopen-hip rocm-cmake hipsparse rocblas-dev rocm-opencl-dev + diff --git a/rvs/conf/MI300X/levels/rvs_level_2.conf b/rvs/conf/MI300X/levels/rvs_level_2.conf new file mode 100644 index 000000000..fb3e444cb --- /dev/null +++ b/rvs/conf/MI300X/levels/rvs_level_2.conf @@ -0,0 +1,75 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: pcie_capability + device: all + module: peqt + capability: + link_cap_max_speed: + link_cap_max_width: + link_stat_cur_speed: + link_stat_neg_width: + slot_pwr_limit_value: + slot_physical_num: + deviceid: + vendor_id: + kernel_driver: + +- name: hbm_basic + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 268435456 + test_type: 1 + mibibytes: true + o/p_csv: false + read: true + write: true + +- name: xgmi_interface + device: all + module: pbqt + log_interval: 5000 + duration: 20000 + peers: all + test_bandwidth: true + bidirectional: false + parallel: true + block_size: 67108864 + device_id: all + +- name: pcie_interface + device: all + module: pebb + duration: 20000 + device_to_host: false + host_to_device: true + parallel: true + block_size: 67108864 + link_type: 2 + diff --git a/rvs/conf/MI300X/levels/rvs_level_3.conf b/rvs/conf/MI300X/levels/rvs_level_3.conf new file mode 100644 index 000000000..ddb35de8e --- /dev/null +++ b/rvs/conf/MI300X/levels/rvs_level_3.conf @@ -0,0 +1,150 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: hbm_mid + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 268435456 + test_type: 1 + mibibytes: true + o/p_csv: false + read: true + write: true + copy: true + add: true + mul: true + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 981000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp8_r + out_data_type: fp8_r + compute_type: fp32_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 523000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 552000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: bf16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + diff --git a/rvs/conf/MI300X/levels/rvs_level_4.conf b/rvs/conf/MI300X/levels/rvs_level_4.conf new file mode 100644 index 000000000..70494f31e --- /dev/null +++ b/rvs/conf/MI300X/levels/rvs_level_4.conf @@ -0,0 +1,229 @@ +# ################################################################################ +# # +# # Copyright (c) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: hbm_full + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 268435456 + test_type: 1 + mibibytes: true + o/p_csv: false + read: true + write: true + copy: true + add: true + mul: true + triad: true + dot: true + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: memtest + device: all + module: mem + parallel: true + count: 1 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 981000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp8_r + out_data_type: fp8_r + compute_type: fp32_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 523000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 552000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: bf16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-fp32-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 100000 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: trig + ops_type: sgemm + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-fp64-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 70000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: trig + ops_type: dgemm + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: power-stress + device: all + module: iet + parallel: true + duration: 600000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + target_power: 750 + tolerance: 0.01 + matrix_size: 28000 + ops_type: dgemm + lda: 28000 + ldb: 28000 + ldc: 28000 + alpha: 1 + beta: 1 + matrix_init: hiprand + diff --git a/rvs/conf/MI300X/levels/rvs_level_5.conf b/rvs/conf/MI300X/levels/rvs_level_5.conf new file mode 100644 index 000000000..ec6fae0e0 --- /dev/null +++ b/rvs/conf/MI300X/levels/rvs_level_5.conf @@ -0,0 +1,229 @@ +# ################################################################################ +# # +# # Copyright (c) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: hbm_full + device: all + module: babel + parallel: true + count: 1 + num_iter: 10000 + array_size: 268435456 + test_type: 1 + mibibytes: true + o/p_csv: false + read: true + write: true + copy: true + add: true + mul: true + triad: true + dot: true + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 900000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: memtest + device: all + module: mem + parallel: true + count: 20 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 981000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp8_r + out_data_type: fp8_r + compute_type: fp32_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 523000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 552000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: bf16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-fp32-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 100000 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: trig + ops_type: sgemm + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-fp64-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 70000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: trig + ops_type: dgemm + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: power-stress + device: all + module: iet + parallel: true + duration: 3600000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + target_power: 750 + tolerance: 0.01 + matrix_size: 28000 + ops_type: dgemm + lda: 28000 + ldb: 28000 + ldc: 28000 + alpha: 1 + beta: 1 + matrix_init: hiprand + diff --git a/rvs/conf/MI308X-HF/babel.conf b/rvs/conf/MI308X-HF/babel.conf index 0aac30965..89daa98bb 100644 --- a/rvs/conf/MI308X-HF/babel.conf +++ b/rvs/conf/MI308X-HF/babel.conf @@ -1,6 +1,6 @@ # ################################################################################ # # -# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # Copyright (c) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. # # # # MIT LICENSE: # # Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -41,11 +41,16 @@ actions: parallel: false # Parallel true or false count: 1 # Number of times you want to repeat the test from the begin ( A clean start every time) num_iter: 3000 # Number of iterations, this many kernels are launched simultaneosuly and stresses the system + duration: 0 # Duration in milliseconds. When > 0, runs for this time instead of num_iter. 0 = use num_iter array_size: 865075200 # Buffer size the test operates, this is 825 MiB test_type: 2 # type of test, 1: Float, 2: Double, 3: Triad float, 4: Triad double mibibytes: false # mibibytes (MiB) or megabytes (MB), true for MiB o/p_csv: false # o/p as csv file - subtest: 5 # 1: copy 2: copy+mul 3: copy+mul+add 4: copy+mul+add+traid 5: copy+mul+add+traid+dot dwords_per_lane: 4 # Number of dwords per lane chunks_per_block: 4 # Number of chunks per block + copy: true # copy operation + mul: true # mul operation + add: true # add operation + dot: true # dot operation + triad: true # triad operation diff --git a/rvs/conf/MI308X-HF/levels/rvs_level_1.conf b/rvs/conf/MI308X-HF/levels/rvs_level_1.conf new file mode 100644 index 000000000..0d3512230 --- /dev/null +++ b/rvs/conf/MI308X-HF/levels/rvs_level_1.conf @@ -0,0 +1,38 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: + +- name: metapackage-validation + device: all + module: rcqt + package: rocm-hip-sdk rocm-hip-libraries rocm-ml-libraries rocm-ml-sdk amdgpu-dkms rocm-language-runtime rocm-openmp-sdk rocm-utils rocm-opencl-sdk rocm-opencl-runtime rocm-hip-runtime rocm-developer-tools + +- name: packagelist-install-validation + device: all + module: rcqt + rpmpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-devel rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-devel hipcub-devel rocsolver rocsparse rocsolver-devel rocminfo hipfft-devel rocm-gdb rocm-dbgapi rocfft hipblas-devel rocthrust-devel openmp-extras-runtime openmp-extras-devel comgr rccl rocblas hipblas roctracer-devel hip-doc rocrand hsa-rocr hipfft hipsparse-devel rocsparse-devel rocrand-devel rocm-opencl hip-devel rocprim-devel hipsolver-devel rocfft-devel hsa-amd-aqlprofile hipify-clang miopen-hip-devel rocm-llvm hip-runtime-amd hip-samples rocalution-devel rccl-devel hipsolver rocprofiler-devel miopen-hip rocm-cmake hipsparse rocblas-devel rocm-opencl-devel + debpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-dev rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-dev hipcub-dev rocsolver rocsparse rocsolver-dev rocminfo hipfft-dev rocm-gdb rocm-dbgapi rocfft hipblas-dev rocthrust-dev openmp-extras-runtime openmp-extras-dev comgr rccl rocblas hipblas roctracer-dev hip-doc rocrand hsa-rocr hipfft hipsparse-dev rocsparse-dev rocrand-dev rocm-opencl hip-dev rocprim-dev hipsolver-dev rocfft-dev hsa-amd-aqlprofile hipify-clang miopen-hip-dev rocm-llvm hip-runtime-amd hip-samples rocalution-dev rccl-dev hipsolver rocprofiler-dev miopen-hip rocm-cmake hipsparse rocblas-dev rocm-opencl-dev + diff --git a/rvs/conf/MI308X-HF/levels/rvs_level_2.conf b/rvs/conf/MI308X-HF/levels/rvs_level_2.conf new file mode 100644 index 000000000..535db6e0a --- /dev/null +++ b/rvs/conf/MI308X-HF/levels/rvs_level_2.conf @@ -0,0 +1,77 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: pcie_capability + device: all + module: peqt + capability: + link_cap_max_speed: + link_cap_max_width: + link_stat_cur_speed: + link_stat_neg_width: + slot_pwr_limit_value: + slot_physical_num: + deviceid: + vendor_id: + kernel_driver: + +- name: hbm_basic + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 865075200 + test_type: 2 + mibibytes: true + o/p_csv: false + read: true + write: true + dwords_per_lane: 4 + chunks_per_block: 4 + +- name: xgmi_interface + device: all + module: pbqt + log_interval: 5000 + duration: 20000 + peers: all + test_bandwidth: true + bidirectional: false + parallel: true + block_size: 67108864 + device_id: all + +- name: pcie_interface + device: all + module: pebb + duration: 20000 + device_to_host: false + host_to_device: true + parallel: true + block_size: 67108864 + link_type: 2 + diff --git a/rvs/conf/MI308X-HF/levels/rvs_level_3.conf b/rvs/conf/MI308X-HF/levels/rvs_level_3.conf new file mode 100644 index 000000000..25998ba93 --- /dev/null +++ b/rvs/conf/MI308X-HF/levels/rvs_level_3.conf @@ -0,0 +1,173 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: hbm_mid + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 865075200 + test_type: 2 + mibibytes: true + o/p_csv: false + read: true + write: true + copy: true + mul: true + add: true + dwords_per_lane: 4 + chunks_per_block: 4 + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 170000 + copy_matrix: false + target_stress: 384000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp8_r + out_data_type: fp8_r + compute_type: fp32_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-i8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 500 + copy_matrix: false + target_stress: 406000 + matrix_size_a: 8192 + matrix_size_b: 13312 + matrix_size_c: 17792 + matrix_init: trig + data_type: i8_r + compute_type: i32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 90000 + copy_matrix: false + target_stress: 188000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 90000 + copy_matrix: false + target_stress: 194000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: bf16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + diff --git a/rvs/conf/MI308X-HF/levels/rvs_level_4.conf b/rvs/conf/MI308X-HF/levels/rvs_level_4.conf new file mode 100644 index 000000000..20e49f116 --- /dev/null +++ b/rvs/conf/MI308X-HF/levels/rvs_level_4.conf @@ -0,0 +1,246 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: hbm_full + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 865075200 + test_type: 2 + mibibytes: true + o/p_csv: false + read: true + write: true + copy: true + mul: true + add: true + triad: true + dot: true + dwords_per_lane: 4 + chunks_per_block: 4 + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: memtest + device: all + module: mem + parallel: true + count: 1 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 170000 + copy_matrix: false + target_stress: 384000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp8_r + out_data_type: fp8_r + compute_type: fp32_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-i8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 500 + copy_matrix: false + target_stress: 406000 + matrix_size_a: 8192 + matrix_size_b: 13312 + matrix_size_c: 17792 + matrix_init: trig + data_type: i8_r + compute_type: i32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 90000 + copy_matrix: false + target_stress: 188000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 90000 + copy_matrix: false + target_stress: 194000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: bf16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-tf32-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 50 + copy_matrix: false + target_stress: 96000 + matrix_size_a: 8192 + matrix_size_b: 12288 + matrix_size_c: 4096 + matrix_init: trig + data_type: fp32_r + compute_type: xf32_r + transa: 0 + transb: 0 + alpha: 1 + beta: 1 + blas_source: hipblaslt + parallel: true + +- name: compute-fp32-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 100 + copy_matrix: false + target_stress: 26000 + matrix_size_a: 8192 + matrix_size_b: 8960 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp32_r + compute_type: fp32_r + transa: 0 + transb: 0 + alpha: 1 + beta: 1 + blas_source: hipblaslt + parallel: true + +- name: power-stress + device: all + module: iet + parallel: true + duration: 600000 + ramp_interval: 1000 + sample_interval: 5000 + log_interval: 5000 + target_power: 650 + tolerance: 0.05 + bw_workload: true + cp_workload: false + wg_count: 64 + diff --git a/rvs/conf/MI308X-HF/levels/rvs_level_5.conf b/rvs/conf/MI308X-HF/levels/rvs_level_5.conf new file mode 100644 index 000000000..010880284 --- /dev/null +++ b/rvs/conf/MI308X-HF/levels/rvs_level_5.conf @@ -0,0 +1,246 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: hbm_full + device: all + module: babel + parallel: true + count: 1 + num_iter: 10000 + array_size: 865075200 + test_type: 2 + mibibytes: true + o/p_csv: false + read: true + write: true + copy: true + mul: true + add: true + triad: true + dot: true + dwords_per_lane: 4 + chunks_per_block: 4 + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 900000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: memtest + device: all + module: mem + parallel: true + count: 20 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 170000 + copy_matrix: false + target_stress: 336441 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp8_r + out_data_type: fp8_r + compute_type: fp32_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-i8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 500 + copy_matrix: false + target_stress: 406000 + matrix_size_a: 8192 + matrix_size_b: 13312 + matrix_size_c: 17792 + matrix_init: trig + data_type: i8_r + compute_type: i32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 90000 + copy_matrix: false + target_stress: 172333 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 90000 + copy_matrix: false + target_stress: 172333 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: bf16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-tf32-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 50 + copy_matrix: false + target_stress: 96000 + matrix_size_a: 8192 + matrix_size_b: 12288 + matrix_size_c: 4096 + matrix_init: trig + data_type: fp32_r + compute_type: xf32_r + transa: 0 + transb: 0 + alpha: 1 + beta: 1 + blas_source: hipblaslt + parallel: true + +- name: compute-fp32-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 100 + copy_matrix: false + target_stress: 26000 + matrix_size_a: 8192 + matrix_size_b: 8960 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp32_r + compute_type: fp32_r + transa: 0 + transb: 0 + alpha: 1 + beta: 1 + blas_source: hipblaslt + parallel: true + +- name: power-stress + device: all + module: iet + parallel: true + duration: 3600000 + ramp_interval: 1000 + sample_interval: 5000 + log_interval: 5000 + target_power: 650 + tolerance: 0.05 + bw_workload: true + cp_workload: false + + diff --git a/rvs/conf/MI308X/babel.conf b/rvs/conf/MI308X/babel.conf index 2af387d5f..821ddbb59 100644 --- a/rvs/conf/MI308X/babel.conf +++ b/rvs/conf/MI308X/babel.conf @@ -1,6 +1,6 @@ # ################################################################################ # # -# # Copyright (c) 2018-2025 Advanced Micro Devices, Inc. All rights reserved. +# # Copyright (c) 2024-2026 Advanced Micro Devices, Inc. All rights reserved. # # # # MIT LICENSE: # # Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -41,11 +41,16 @@ actions: parallel: false # Parallel true or false count: 1 # Number of times you want to repeat the test from the begin ( A clean start every time) num_iter: 3000 # Number of iterations, this many kernels are launched simultaneosuly and stresses the system + duration: 0 # Duration in milliseconds. When > 0, runs for this time instead of num_iter. 0 = use num_iter array_size: 865075200 # Buffer size the test operates, this is 825 MiB test_type: 2 # type of test, 1: Float, 2: Double, 3: Triad float, 4: Triad double mibibytes: false # mibibytes (MiB) or megabytes (MB), true for MiB o/p_csv: false # o/p as csv file - subtest: 5 # 1: copy 2: copy+mul 3: copy+mul+add 4: copy+mul+add+traid 5: copy+mul+add+traid+dot dwords_per_lane: 4 # Number of dwords per lane chunks_per_block: 4 # Number of chunks per block + copy: true # copy operation + mul: true # mul operation + add: true # add operation + dot: true # dot operation + triad: true # triad operation diff --git a/rvs/conf/MI308X/levels/rvs_level_1.conf b/rvs/conf/MI308X/levels/rvs_level_1.conf new file mode 100644 index 000000000..0d3512230 --- /dev/null +++ b/rvs/conf/MI308X/levels/rvs_level_1.conf @@ -0,0 +1,38 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: + +- name: metapackage-validation + device: all + module: rcqt + package: rocm-hip-sdk rocm-hip-libraries rocm-ml-libraries rocm-ml-sdk amdgpu-dkms rocm-language-runtime rocm-openmp-sdk rocm-utils rocm-opencl-sdk rocm-opencl-runtime rocm-hip-runtime rocm-developer-tools + +- name: packagelist-install-validation + device: all + module: rcqt + rpmpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-devel rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-devel hipcub-devel rocsolver rocsparse rocsolver-devel rocminfo hipfft-devel rocm-gdb rocm-dbgapi rocfft hipblas-devel rocthrust-devel openmp-extras-runtime openmp-extras-devel comgr rccl rocblas hipblas roctracer-devel hip-doc rocrand hsa-rocr hipfft hipsparse-devel rocsparse-devel rocrand-devel rocm-opencl hip-devel rocprim-devel hipsolver-devel rocfft-devel hsa-amd-aqlprofile hipify-clang miopen-hip-devel rocm-llvm hip-runtime-amd hip-samples rocalution-devel rccl-devel hipsolver rocprofiler-devel miopen-hip rocm-cmake hipsparse rocblas-devel rocm-opencl-devel + debpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-dev rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-dev hipcub-dev rocsolver rocsparse rocsolver-dev rocminfo hipfft-dev rocm-gdb rocm-dbgapi rocfft hipblas-dev rocthrust-dev openmp-extras-runtime openmp-extras-dev comgr rccl rocblas hipblas roctracer-dev hip-doc rocrand hsa-rocr hipfft hipsparse-dev rocsparse-dev rocrand-dev rocm-opencl hip-dev rocprim-dev hipsolver-dev rocfft-dev hsa-amd-aqlprofile hipify-clang miopen-hip-dev rocm-llvm hip-runtime-amd hip-samples rocalution-dev rccl-dev hipsolver rocprofiler-dev miopen-hip rocm-cmake hipsparse rocblas-dev rocm-opencl-dev + diff --git a/rvs/conf/MI308X/levels/rvs_level_2.conf b/rvs/conf/MI308X/levels/rvs_level_2.conf new file mode 100644 index 000000000..535db6e0a --- /dev/null +++ b/rvs/conf/MI308X/levels/rvs_level_2.conf @@ -0,0 +1,77 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: pcie_capability + device: all + module: peqt + capability: + link_cap_max_speed: + link_cap_max_width: + link_stat_cur_speed: + link_stat_neg_width: + slot_pwr_limit_value: + slot_physical_num: + deviceid: + vendor_id: + kernel_driver: + +- name: hbm_basic + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 865075200 + test_type: 2 + mibibytes: true + o/p_csv: false + read: true + write: true + dwords_per_lane: 4 + chunks_per_block: 4 + +- name: xgmi_interface + device: all + module: pbqt + log_interval: 5000 + duration: 20000 + peers: all + test_bandwidth: true + bidirectional: false + parallel: true + block_size: 67108864 + device_id: all + +- name: pcie_interface + device: all + module: pebb + duration: 20000 + device_to_host: false + host_to_device: true + parallel: true + block_size: 67108864 + link_type: 2 + diff --git a/rvs/conf/MI308X/levels/rvs_level_3.conf b/rvs/conf/MI308X/levels/rvs_level_3.conf new file mode 100644 index 000000000..912352a0b --- /dev/null +++ b/rvs/conf/MI308X/levels/rvs_level_3.conf @@ -0,0 +1,173 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: hbm_mid + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 865075200 + test_type: 2 + mibibytes: true + o/p_csv: false + read: true + write: true + copy: true + mul: true + add: true + dwords_per_lane: 4 + chunks_per_block: 4 + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 170000 + copy_matrix: false + target_stress: 336441 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp8_r + out_data_type: fp8_r + compute_type: fp32_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-i8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 500 + copy_matrix: false + target_stress: 406000 + matrix_size_a: 8192 + matrix_size_b: 13312 + matrix_size_c: 17792 + matrix_init: trig + data_type: i8_r + compute_type: i32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 90000 + copy_matrix: false + target_stress: 172333 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 90000 + copy_matrix: false + target_stress: 172333 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: bf16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + diff --git a/rvs/conf/MI308X/levels/rvs_level_4.conf b/rvs/conf/MI308X/levels/rvs_level_4.conf new file mode 100644 index 000000000..fe862c18e --- /dev/null +++ b/rvs/conf/MI308X/levels/rvs_level_4.conf @@ -0,0 +1,245 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: hbm_full + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 865075200 + test_type: 2 + mibibytes: true + o/p_csv: false + read: true + write: true + copy: true + mul: true + add: true + triad: true + dot: true + dwords_per_lane: 4 + chunks_per_block: 4 + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: memtest + device: all + module: mem + parallel: true + count: 1 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 170000 + copy_matrix: false + target_stress: 336441 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp8_r + out_data_type: fp8_r + compute_type: fp32_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-i8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 500 + copy_matrix: false + target_stress: 406000 + matrix_size_a: 8192 + matrix_size_b: 13312 + matrix_size_c: 17792 + matrix_init: trig + data_type: i8_r + compute_type: i32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 90000 + copy_matrix: false + target_stress: 172333 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 90000 + copy_matrix: false + target_stress: 172333 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: bf16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-tf32-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 50 + copy_matrix: false + target_stress: 96000 + matrix_size_a: 8192 + matrix_size_b: 12288 + matrix_size_c: 4096 + matrix_init: trig + data_type: fp32_r + compute_type: xf32_r + transa: 0 + transb: 0 + alpha: 1 + beta: 1 + blas_source: hipblaslt + parallel: true + +- name: compute-fp32-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 100 + copy_matrix: false + target_stress: 26000 + matrix_size_a: 8192 + matrix_size_b: 8960 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp32_r + compute_type: fp32_r + transa: 0 + transb: 0 + alpha: 1 + beta: 1 + blas_source: hipblaslt + parallel: true + +- name: power-stress + device: all + module: iet + parallel: true + duration: 600000 + ramp_interval: 1000 + sample_interval: 5000 + log_interval: 5000 + target_power: 650 + tolerance: 0.05 + bw_workload: true + cp_workload: false + diff --git a/rvs/conf/MI308X/levels/rvs_level_5.conf b/rvs/conf/MI308X/levels/rvs_level_5.conf new file mode 100644 index 000000000..010880284 --- /dev/null +++ b/rvs/conf/MI308X/levels/rvs_level_5.conf @@ -0,0 +1,246 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: hbm_full + device: all + module: babel + parallel: true + count: 1 + num_iter: 10000 + array_size: 865075200 + test_type: 2 + mibibytes: true + o/p_csv: false + read: true + write: true + copy: true + mul: true + add: true + triad: true + dot: true + dwords_per_lane: 4 + chunks_per_block: 4 + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 900000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: memtest + device: all + module: mem + parallel: true + count: 20 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 170000 + copy_matrix: false + target_stress: 336441 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp8_r + out_data_type: fp8_r + compute_type: fp32_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-i8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 500 + copy_matrix: false + target_stress: 406000 + matrix_size_a: 8192 + matrix_size_b: 13312 + matrix_size_c: 17792 + matrix_init: trig + data_type: i8_r + compute_type: i32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 90000 + copy_matrix: false + target_stress: 172333 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 90000 + copy_matrix: false + target_stress: 172333 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: bf16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-tf32-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 50 + copy_matrix: false + target_stress: 96000 + matrix_size_a: 8192 + matrix_size_b: 12288 + matrix_size_c: 4096 + matrix_init: trig + data_type: fp32_r + compute_type: xf32_r + transa: 0 + transb: 0 + alpha: 1 + beta: 1 + blas_source: hipblaslt + parallel: true + +- name: compute-fp32-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 100 + copy_matrix: false + target_stress: 26000 + matrix_size_a: 8192 + matrix_size_b: 8960 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp32_r + compute_type: fp32_r + transa: 0 + transb: 0 + alpha: 1 + beta: 1 + blas_source: hipblaslt + parallel: true + +- name: power-stress + device: all + module: iet + parallel: true + duration: 3600000 + ramp_interval: 1000 + sample_interval: 5000 + log_interval: 5000 + target_power: 650 + tolerance: 0.05 + bw_workload: true + cp_workload: false + + diff --git a/rvs/conf/MI325X/babel.conf b/rvs/conf/MI325X/babel.conf index 55751ab38..d2a32b047 100644 --- a/rvs/conf/MI325X/babel.conf +++ b/rvs/conf/MI325X/babel.conf @@ -1,6 +1,6 @@ # ################################################################################ # # -# # Copyright (c) 2024 Advanced Micro Devices, Inc. All rights reserved. +# # Copyright (c) 2024-2026 Advanced Micro Devices, Inc. All rights reserved. # # # # MIT LICENSE: # # Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -41,9 +41,16 @@ actions: parallel: false # Parallel true or false count: 1 # Number of times you want to repeat the test from the begin ( A clean start every time) num_iter: 5000 # Number of iterations, this many kernels are launched simultaneosuly and stresses the system + duration: 0 # Duration in milliseconds. When > 0, runs for this time instead of num_iter. 0 = use num_iter array_size: 268435456 # Buffer size the test operates, this is 256 MiB test_type: 1 # type of test, 1: Float, 2: Double, 3: Triad float, 4: Triad double mibibytes: true # mibibytes (MiB) or megabytes (MB), true for MiB o/p_csv: false # o/p as csv file - subtest: 5 # 1: copy 2: copy+mul 3: copy+mul+add 4: copy+mul+add+traid 5: copy+mul+add+traid+dot + read: true # read operation + write: true # write operation + copy: true # copy operation + mul: true # mul operation + add: true # add operation + dot: true # dot operation + triad: true # triad operation diff --git a/rvs/conf/MI325X/iet_stress.conf b/rvs/conf/MI325X/iet_stress.conf index d92bca9e6..bef3adf7e 100644 --- a/rvs/conf/MI325X/iet_stress.conf +++ b/rvs/conf/MI325X/iet_stress.conf @@ -1,6 +1,6 @@ # ################################################################################ # # -# # Copyright (c) 2024 Advanced Micro Devices, Inc. All rights reserved. +# # Copyright (c) 2024-2026 Advanced Micro Devices, Inc. All rights reserved. # # # # MIT LICENSE: # # Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -34,6 +34,7 @@ # Set matrix_size to 28000. # Test duration set to 10 mins. # Target power set to 1000W for each GPU. +# Power tolerance of 1% (10W). # # Run test with: # cd bin @@ -53,6 +54,7 @@ actions: sample_interval: 5000 log_interval: 5000 target_power: 1000 + tolerance: 0.01 matrix_size: 28000 ops_type: dgemm lda: 28000 diff --git a/rvs/conf/MI325X/levels/rvs_level_1.conf b/rvs/conf/MI325X/levels/rvs_level_1.conf new file mode 100644 index 000000000..0d3512230 --- /dev/null +++ b/rvs/conf/MI325X/levels/rvs_level_1.conf @@ -0,0 +1,38 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: + +- name: metapackage-validation + device: all + module: rcqt + package: rocm-hip-sdk rocm-hip-libraries rocm-ml-libraries rocm-ml-sdk amdgpu-dkms rocm-language-runtime rocm-openmp-sdk rocm-utils rocm-opencl-sdk rocm-opencl-runtime rocm-hip-runtime rocm-developer-tools + +- name: packagelist-install-validation + device: all + module: rcqt + rpmpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-devel rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-devel hipcub-devel rocsolver rocsparse rocsolver-devel rocminfo hipfft-devel rocm-gdb rocm-dbgapi rocfft hipblas-devel rocthrust-devel openmp-extras-runtime openmp-extras-devel comgr rccl rocblas hipblas roctracer-devel hip-doc rocrand hsa-rocr hipfft hipsparse-devel rocsparse-devel rocrand-devel rocm-opencl hip-devel rocprim-devel hipsolver-devel rocfft-devel hsa-amd-aqlprofile hipify-clang miopen-hip-devel rocm-llvm hip-runtime-amd hip-samples rocalution-devel rccl-devel hipsolver rocprofiler-devel miopen-hip rocm-cmake hipsparse rocblas-devel rocm-opencl-devel + debpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-dev rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-dev hipcub-dev rocsolver rocsparse rocsolver-dev rocminfo hipfft-dev rocm-gdb rocm-dbgapi rocfft hipblas-dev rocthrust-dev openmp-extras-runtime openmp-extras-dev comgr rccl rocblas hipblas roctracer-dev hip-doc rocrand hsa-rocr hipfft hipsparse-dev rocsparse-dev rocrand-dev rocm-opencl hip-dev rocprim-dev hipsolver-dev rocfft-dev hsa-amd-aqlprofile hipify-clang miopen-hip-dev rocm-llvm hip-runtime-amd hip-samples rocalution-dev rccl-dev hipsolver rocprofiler-dev miopen-hip rocm-cmake hipsparse rocblas-dev rocm-opencl-dev + diff --git a/rvs/conf/MI325X/levels/rvs_level_2.conf b/rvs/conf/MI325X/levels/rvs_level_2.conf new file mode 100644 index 000000000..fb3e444cb --- /dev/null +++ b/rvs/conf/MI325X/levels/rvs_level_2.conf @@ -0,0 +1,75 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: pcie_capability + device: all + module: peqt + capability: + link_cap_max_speed: + link_cap_max_width: + link_stat_cur_speed: + link_stat_neg_width: + slot_pwr_limit_value: + slot_physical_num: + deviceid: + vendor_id: + kernel_driver: + +- name: hbm_basic + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 268435456 + test_type: 1 + mibibytes: true + o/p_csv: false + read: true + write: true + +- name: xgmi_interface + device: all + module: pbqt + log_interval: 5000 + duration: 20000 + peers: all + test_bandwidth: true + bidirectional: false + parallel: true + block_size: 67108864 + device_id: all + +- name: pcie_interface + device: all + module: pebb + duration: 20000 + device_to_host: false + host_to_device: true + parallel: true + block_size: 67108864 + link_type: 2 + diff --git a/rvs/conf/MI325X/levels/rvs_level_3.conf b/rvs/conf/MI325X/levels/rvs_level_3.conf new file mode 100644 index 000000000..37ff5767e --- /dev/null +++ b/rvs/conf/MI325X/levels/rvs_level_3.conf @@ -0,0 +1,150 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: hbm_mid + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 268435456 + test_type: 1 + mibibytes: true + o/p_csv: false + read: true + write: true + copy: true + add: true + mul: true + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 983000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp8_r + out_data_type: fp8_r + compute_type: fp32_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 524000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 554000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: bf16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + diff --git a/rvs/conf/MI325X/levels/rvs_level_4.conf b/rvs/conf/MI325X/levels/rvs_level_4.conf new file mode 100644 index 000000000..4e71ebccc --- /dev/null +++ b/rvs/conf/MI325X/levels/rvs_level_4.conf @@ -0,0 +1,229 @@ +# ################################################################################ +# # +# # Copyright (c) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: hbm_full + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 268435456 + test_type: 1 + mibibytes: true + o/p_csv: false + read: true + write: true + copy: true + add: true + mul: true + triad: true + dot: true + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: memtest + device: all + module: mem + parallel: true + count: 1 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 983000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp8_r + out_data_type: fp8_r + compute_type: fp32_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 524000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 554000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: bf16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-fp32-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 100000 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: trig + ops_type: sgemm + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-fp64-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 70000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: trig + ops_type: dgemm + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: power-stress + device: all + module: iet + parallel: true + duration: 60000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + target_power: 1000 + tolerance: 0.01 + matrix_size: 28000 + ops_type: dgemm + lda: 28000 + ldb: 28000 + ldc: 28000 + alpha: 1 + beta: 1 + matrix_init: hiprand + diff --git a/rvs/conf/MI325X/levels/rvs_level_5.conf b/rvs/conf/MI325X/levels/rvs_level_5.conf new file mode 100644 index 000000000..7883c94cc --- /dev/null +++ b/rvs/conf/MI325X/levels/rvs_level_5.conf @@ -0,0 +1,229 @@ +# ################################################################################ +# # +# # Copyright (c) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: hbm_full + device: all + module: babel + parallel: true + count: 1 + num_iter: 10000 + array_size: 268435456 + test_type: 1 + mibibytes: true + o/p_csv: false + read: true + write: true + copy: true + add: true + mul: true + triad: true + dot: true + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 900000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: memtest + device: all + module: mem + parallel: true + count: 20 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 983000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp8_r + out_data_type: fp8_r + compute_type: fp32_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 524000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 554000 + matrix_size_a: 4864 + matrix_size_b: 4096 + matrix_size_c: 8192 + matrix_init: trig + data_type: bf16_r + lda: 8320 + ldb: 8320 + ldc: 4992 + ldd: 4992 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-fp32-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 100000 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: trig + ops_type: sgemm + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: compute-fp64-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 70000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: trig + ops_type: dgemm + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + parallel: true + +- name: power-stress + device: all + module: iet + parallel: true + duration: 3600000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + target_power: 1000 + tolerance: 0.01 + matrix_size: 28000 + ops_type: dgemm + lda: 28000 + ldb: 28000 + ldc: 28000 + alpha: 1 + beta: 1 + matrix_init: hiprand + diff --git a/rvs/conf/gm_single.conf b/rvs/conf/MI350P-450W/babel_single.conf similarity index 68% rename from rvs/conf/gm_single.conf rename to rvs/conf/MI350P-450W/babel_single.conf index aaf70da20..165e7c22d 100644 --- a/rvs/conf/gm_single.conf +++ b/rvs/conf/MI350P-450W/babel_single.conf @@ -1,6 +1,6 @@ # ################################################################################ # # -# # Copyright (c) 2018-2023 Advanced Micro Devices, Inc. All rights reserved. +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. # # # # MIT LICENSE: # # Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -22,34 +22,34 @@ # # SOFTWARE. # # # ############################################################################### - -# GM test +# BABEL test # # Preconditions: # Set device to all. If you need to run the rvs only on a subset of GPUs, please run rvs with -g # option, collect the GPUs IDs (e.g.: GPU[ 5 - 50599] -> 50599 is the GPU ID) and then specify -# all the GPUs IDs separated by white space. -# Set monitor to true for monitoring to start. -# -# Run test with: -# cd bin -# sudo ./rvs -c conf/gm_single.conf -d 3 -# -# Expected result: -# Monitor metrics for 30 seconds and display average values for junction temperature, -# memory clock, system clock, fan % and power for each GPU on the machine. Also display -# bound value violations during the monitor duration. +# all the GPUs IDs separated by white space (e.g.: device: 50599 3245) +# Set parallel execution to false +# Set run count to 1 (test will run once) # + actions: -- name: metrics_monitor - module: gm +- name: babel-double-4gib device: all - monitor: true - metrics: - temp: true 100 0 - fan: true 100 0 - mem_clock: true 1000 0 - clock: true 1000 0 - power: true 750 0 - duration: 10000 - + module: babel + parallel: false + count: 1 + num_iter: 3000 + array_size: 536870912 + test_type: 2 + mibibytes: true + o/p_csv: false + read: true + write: true + copy: true + mul: true + add: true + dot: true + triad: true + dwords_per_lane: 4 + chunks_per_block: 1 + tb_size: 1024 diff --git a/rvs/conf/MI350P-450W/gst_single.conf b/rvs/conf/MI350P-450W/gst_single.conf new file mode 100644 index 000000000..0827963b8 --- /dev/null +++ b/rvs/conf/MI350P-450W/gst_single.conf @@ -0,0 +1,200 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# GST test - gst-96Tflops-8K12K4K-trig-tf32 +# +# Preconditions: +# Set device to all. If you need to run the rvs only on a subset of GPUs, please run rvs with -g +# option, collect the GPUs IDs (e.g.: GPU[ 5 - 50599] -> 50599 is the GPU ID) and then specify +# all the GPUs IDs separated by white space +# Set matrices sizes to 8192 * 12288 * 4096 +# Set matrix data type as fp32 real number +# Set compute type as tf32 (xf32) +# Set matrix data initialization method as trignometric float +# Set copy_matrix to false (the matrices will be copied to GPUs only once) +# Set target stress GFLOPS as 96 TFLOPS +# Set blas source (backend) as hipblaslt +# +# Expected result: +# The test on each GPU passes (TRUE) if the GPU achieves 96 TFLOPS or more +# within the test duration of 15 seconds after ramp-up duration of 5 seconds. +# Else test on the GPU fails (FALSE). + +actions: +- name: gst-8K8K16K-trig-bf16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-8K8K16K-trig-fp16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-8K8K16K-trig-fp8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-4K4K32K-trig-fp4 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 32768 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp4_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + +- name: gst-Tflops-4K4K16K-trig-bf8 + module: gst + device: all + matrix_init: trig + target_stress: 0 + data_type: fp8_e5m2_r # BF8 (OCP E5M2) -> HIP_R_8F_E5M2 + out_data_type: bf16_r + compute_type: fp32_r + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 16384 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + duration: 15000 + hot_calls: 1000 + blas_source: hipblaslt + +- name: gst-Tflops-4K4K32K-trig-fp6 + module: gst + device: all + target_stress: 0 + matrix_init: trig + data_type: fp6_e3m2_r # MX FP6 E3M2 -> HIP_R_6F_E3M2 + out_data_type: fp16_r + compute_type: fp32_r + scale_a: block + scale_b: block + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 32768 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + duration: 15000 + blas_source: hipblaslt + +- name: gst-8K8K16K-rand-fp8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: rand + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + diff --git a/rvs/conf/MI350P-450W/iet_stress.conf b/rvs/conf/MI350P-450W/iet_stress.conf new file mode 100644 index 000000000..ff32405a3 --- /dev/null +++ b/rvs/conf/MI350P-450W/iet_stress.conf @@ -0,0 +1,61 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# IET stress test - MI350P long running power validation using typed GEMM via hipBLASLt +# +# Preconditions: +# Set device to all. If you need to run the rvs only on a subset of GPUs, please run rvs with -g +# option, collect the GPUs IDs (e.g.: GPU[ 5 - 50599] -> 50599 is the GPU ID) and then specify +# all the GPUs IDs separated by white space +# Set blas source (backend) as hipblaslt +# Set target power according to MI350P TDP +# +# Run test with: +# cd bin +# ./rvs -c conf/MI350P/iet_stress.conf +# +# Expected result: +# The test on each GPU passes (TRUE) if the GPU power reaches at least +# the target_power threshold, FALSE otherwise + +actions: +- name: iet-stress-450W-dgemm-true + device: all + module: iet + parallel: true + duration: 600000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + target_power: 438 + tolerance: 0.01 + matrix_size: 8192 + ops_type: dgemm + lda: 8192 + ldb: 8192 + ldc: 8192 + alpha: 1 + beta: 1 + matrix_init: hiprand diff --git a/rvs/conf/MI350P-450W/levels/rvs_level_1.conf b/rvs/conf/MI350P-450W/levels/rvs_level_1.conf new file mode 100644 index 000000000..14b8d6c62 --- /dev/null +++ b/rvs/conf/MI350P-450W/levels/rvs_level_1.conf @@ -0,0 +1,38 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: + +- name: metapackage-validation + device: all + module: rcqt + package: rocm-hip-sdk rocm-hip-libraries rocm-ml-libraries rocm-ml-sdk amdgpu-dkms rocm-language-runtime rocm-openmp-sdk rocm-utils rocm-opencl-sdk rocm-opencl-runtime rocm-hip-runtime rocm-developer-tools + +- name: packagelist-install-validation + device: all + module: rcqt + rpmpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-devel rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-devel hipcub-devel rocsolver rocsparse rocsolver-devel rocminfo hipfft-devel rocm-gdb rocm-dbgapi rocfft hipblas-devel rocthrust-devel openmp-extras-runtime openmp-extras-devel comgr rccl rocblas hipblas roctracer-devel hip-doc rocrand hsa-rocr hipfft hipsparse-devel rocsparse-devel rocrand-devel rocm-opencl hip-devel rocprim-devel hipsolver-devel rocfft-devel hsa-amd-aqlprofile hipify-clang miopen-hip-devel rocm-llvm hip-runtime-amd hip-samples rocalution-devel rccl-devel hipsolver rocprofiler-devel miopen-hip rocm-cmake hipsparse rocblas-devel rocm-opencl-devel + debpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-dev rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-dev hipcub-dev rocsolver rocsparse rocsolver-dev rocminfo hipfft-dev rocm-gdb rocm-dbgapi rocfft hipblas-dev rocthrust-dev openmp-extras-runtime openmp-extras-dev comgr rccl rocblas hipblas roctracer-dev hip-doc rocrand hsa-rocr hipfft hipsparse-dev rocsparse-dev rocrand-dev rocm-opencl hip-dev rocprim-dev hipsolver-dev rocfft-dev hsa-amd-aqlprofile hipify-clang miopen-hip-dev rocm-llvm hip-runtime-amd hip-samples rocalution-dev rccl-dev hipsolver rocprofiler-dev miopen-hip rocm-cmake hipsparse rocblas-dev rocm-opencl-dev + diff --git a/rvs/conf/MI350P-450W/levels/rvs_level_2.conf b/rvs/conf/MI350P-450W/levels/rvs_level_2.conf new file mode 100644 index 000000000..f34ed405f --- /dev/null +++ b/rvs/conf/MI350P-450W/levels/rvs_level_2.conf @@ -0,0 +1,79 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: pcie_capability + device: all + module: peqt + capability: + link_cap_max_speed: + link_cap_max_width: + link_stat_cur_speed: + link_stat_neg_width: + slot_pwr_limit_value: + slot_physical_num: + deviceid: + vendor_id: + kernel_driver: + +- name: hbm_basic + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 865075200 + test_type: 2 + mibibytes: false + o/p_csv: false + read: true + write: true + subtest: 0 + dwords_per_lane: 4 + chunks_per_block: 1 + tb_size: 512 + +- name: xgmi_interface + device: all + module: pbqt + log_interval: 5000 + duration: 20000 + peers: all + test_bandwidth: true + bidirectional: false + parallel: true + block_size: 67108864 + device_id: all + +- name: pcie_interface + device: all + module: pebb + duration: 20000 + device_to_host: false + host_to_device: true + parallel: true + block_size: 67108864 + link_type: 2 + diff --git a/rvs/conf/MI350P-450W/levels/rvs_level_3.conf b/rvs/conf/MI350P-450W/levels/rvs_level_3.conf new file mode 100644 index 000000000..4e4ac05d8 --- /dev/null +++ b/rvs/conf/MI350P-450W/levels/rvs_level_3.conf @@ -0,0 +1,252 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# MI350P-450W test level 3 - basic performance +# GST target_stress values are 0 (measurement-only). Calibrated MI350P-450W +# thresholds will be set once they are characterized for this SKU. + +actions: +- name: hbm_mid + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 865075200 + test_type: 2 + mibibytes: false + o/p_csv: false + read: true + write: true + copy: true + mul: true + add: true + dwords_per_lane: 4 + chunks_per_block: 1 + tb_size: 512 + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: compute-fp4-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp4_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e3m2_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-bf6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e2m3_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + diff --git a/rvs/conf/MI350P-450W/levels/rvs_level_4.conf b/rvs/conf/MI350P-450W/levels/rvs_level_4.conf new file mode 100644 index 000000000..196091fdd --- /dev/null +++ b/rvs/conf/MI350P-450W/levels/rvs_level_4.conf @@ -0,0 +1,335 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# MI350P-450W test level 4 - extended performance +# GST target_stress = 0 (measurement-only) until 450W thresholds are calibrated. +# IET target_power = 438 W (per MI350P-450W TDP, matching iet_stress.conf). + +actions: +- name: hbm_full + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 865075200 + test_type: 2 + mibibytes: false + o/p_csv: false + read: true + write: true + copy: true + mul: true + add: true + triad: true + dot: true + dwords_per_lane: 4 + chunks_per_block: 1 + tb_size: 512 + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: memtest + device: all + module: mem + parallel: true + count: 1 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + +- name: compute-fp4-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp4_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e3m2_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-bf6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e2m3_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-fp32-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: trig + data_type: fp32_r + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-fp64-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp64_r + compute_type: fp64_r + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: power-stress + device: all + module: iet + parallel: true + duration: 60000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + target_power: 438 + tolerance: 0.01 + matrix_size: 8192 + ops_type: dgemm + lda: 8192 + ldb: 8192 + ldc: 8192 + alpha: 1 + beta: 1 + matrix_init: hiprand + diff --git a/rvs/conf/MI350P-450W/levels/rvs_level_5.conf b/rvs/conf/MI350P-450W/levels/rvs_level_5.conf new file mode 100644 index 000000000..ddb3e1ece --- /dev/null +++ b/rvs/conf/MI350P-450W/levels/rvs_level_5.conf @@ -0,0 +1,335 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# MI350P-450W test level 5 - full stress test (long duration) +# GST target_stress = 0 (measurement-only) until 450W thresholds are calibrated. +# IET target_power = 438 W (per MI350P-450W TDP). + +actions: +- name: hbm_full + device: all + module: babel + parallel: true + count: 1 + num_iter: 10000 + array_size: 865075200 + test_type: 2 + mibibytes: false + o/p_csv: false + read: true + write: true + copy: true + mul: true + add: true + triad: true + dot: true + dwords_per_lane: 4 + chunks_per_block: 1 + tb_size: 512 + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 900000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: memtest + device: all + module: mem + parallel: true + count: 20 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + +- name: compute-fp4-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp4_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e3m2_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-bf6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e2m3_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-fp32-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: trig + data_type: fp32_r + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-fp64-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp64_r + compute_type: fp64_r + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: power-stress + device: all + module: iet + parallel: true + duration: 3600000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + target_power: 438 + tolerance: 0.01 + matrix_size: 8192 + ops_type: dgemm + lda: 8192 + ldb: 8192 + ldc: 8192 + alpha: 1 + beta: 1 + matrix_init: hiprand + diff --git a/rvs/conf/MI350P-450W/pbqt_single.conf b/rvs/conf/MI350P-450W/pbqt_single.conf new file mode 100644 index 000000000..02f990ea1 --- /dev/null +++ b/rvs/conf/MI350P-450W/pbqt_single.conf @@ -0,0 +1,118 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### +# +# PBQT peer bandwidth tests for AMD Instinct MI350P (PCIe-attached). +# MI350P GPUs are connected via PCIe (typical hop count 2), not XGMI. Do not use +# MI350X configs: TransferBench all-to-all defaults to a2a_direct=1, which only +# includes hop-count-1 (direct XGMI) pairs and yields no valid transfers on MI350P. +# +# Usage: +# ./rvs -c conf/MI350P-450W/pbqt_single.conf +# ./rvs -m pbqt + +actions: +# Peer-to-Peer/Device-to-Device/GPU-to-GPU unidirectional bandwidth test +# using RVS native method via SDMA +- name: pcie_d2d_unidir_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: false + parallel: false + block_size: 1073741824 + device_id: all + +# Peer-to-Peer/Device-to-Device/GPU-to-GPU bidirectional bandwidth test +# using RVS native method via SDMA +- name: pcie_d2d_bidir_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +# Peer-to-Peer/Device-to-Device/GPU-to-GPU unidirectional bandwidth test +# using TransferBench via SDMA +- name: pcie_d2d_unidir_tb_dma + device: all + module: pbqt + duration: 0 + peers: all + test_bandwidth: true + bidirectional: false + parallel: false + block_size: 1073741824 + device_id: all + transfer_method: transferbench + executor: dma + subexecutor: 1 + warm_calls: 2 + hot_calls: 5 + +# Peer-to-Peer/Device-to-Device/GPU-to-GPU unidirectional bandwidth test +# using TransferBench via GFX +- name: pcie_d2d_unidir_tb_gfx + device: all + module: pbqt + duration: 0 + peers: all + test_bandwidth: true + bidirectional: false + parallel: false + block_size: 1073741824 + device_id: all + transfer_method: transferbench + executor: gfx + subexecutor: 16 + warm_calls: 2 + hot_calls: 5 + +# All-to-All bandwidth test using TransferBench via GFX +# a2a_direct: 0 — include PCIe-connected pairs (hop count > 1); default 1 is XGMI-only +- name: pcie_d2d_unidir_tb_a2a + device: all + module: pbqt + peers: all + test_bandwidth: true + bidirectional: false + parallel: false + block_size: 268435456 + device_id: all + transfer_method: transferbench + transferbench_test: alltoall + executor: gfx + subexecutor: 8 + warm_calls: 3 + hot_calls: 10 + gfx_unroll: 2 + a2a_direct: 0 diff --git a/rvs/conf/MI350P-450W/pebb_single.conf b/rvs/conf/MI350P-450W/pebb_single.conf new file mode 100644 index 000000000..1e49a1652 --- /dev/null +++ b/rvs/conf/MI350P-450W/pebb_single.conf @@ -0,0 +1,127 @@ +# ################################################################################ +# # +# # Copyright (c) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### +# +# PEBB bandwidth tests for AMD Instinct MI350P (PCIe-attached). +# Host/device transfers use link_type 2 (PCIe). Do not use MI350X configs on +# MI350P; GPU-to-GPU topology differs (PCIe hop count 2 vs XGMI on MI350X). +# +# Usage: +# ./rvs -c conf/MI350P-450W/pebb_single.conf +# ./rvs -m pebb + +actions: +# Host-to-Device/CPU-to-GPU bandwidth test using RVS native method via SDMA +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +# Device-to-Host/GPU-to-CPU bandwidth test using RVS native method via SDMA +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +# Host-to-Device/CPU-to-GPU bandwidth test using TransferBench via SDMA +- name: pcie_h2d_bandwidth_tb_dma + device: all + module: pebb + duration: 0 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + transfer_method: transferbench + executor: dma + subexecutor: 1 + warm_calls: 3 + hot_calls: 10 + source_memory: cpu + destination_memory: gpu + +# Device-to-Host/GPU-to-CPU bandwidth test using TransferBench via SDMA +- name: pcie_d2h_bandwidth_tb_dma + device: all + module: pebb + duration: 0 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + transfer_method: transferbench + executor: dma + subexecutor: 1 + warm_calls: 3 + hot_calls: 10 + source_memory: gpu + destination_memory: cpu + +# Host-to-Device/CPU-to-GPU bandwidth test using TransferBench via GFX +- name: pcie_h2d_bandwidth_tb_gfx + device: all + module: pebb + duration: 0 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + transfer_method: transferbench + executor: gfx + subexecutor: 8 + warm_calls: 3 + hot_calls: 10 + source_memory: cpu + destination_memory: "null" + +# Device-to-Host/GPU-to-CPU bandwidth test using TransferBench via GFX +- name: pcie_d2h_bandwidth_tb_gfx + device: all + module: pebb + duration: 0 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + transfer_method: transferbench + executor: gfx + subexecutor: 8 + warm_calls: 3 + hot_calls: 10 + source_memory: "null" + destination_memory: cpu diff --git a/rvs/conf/MI350P-600W/babel_single.conf b/rvs/conf/MI350P-600W/babel_single.conf new file mode 100644 index 000000000..165e7c22d --- /dev/null +++ b/rvs/conf/MI350P-600W/babel_single.conf @@ -0,0 +1,55 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### +# BABEL test +# +# Preconditions: +# Set device to all. If you need to run the rvs only on a subset of GPUs, please run rvs with -g +# option, collect the GPUs IDs (e.g.: GPU[ 5 - 50599] -> 50599 is the GPU ID) and then specify +# all the GPUs IDs separated by white space (e.g.: device: 50599 3245) +# Set parallel execution to false +# Set run count to 1 (test will run once) +# + +actions: +- name: babel-double-4gib + device: all + module: babel + parallel: false + count: 1 + num_iter: 3000 + array_size: 536870912 + test_type: 2 + mibibytes: true + o/p_csv: false + read: true + write: true + copy: true + mul: true + add: true + dot: true + triad: true + dwords_per_lane: 4 + chunks_per_block: 1 + tb_size: 1024 diff --git a/rvs/conf/MI350P-600W/gst_single.conf b/rvs/conf/MI350P-600W/gst_single.conf new file mode 100644 index 000000000..ccc49e4bb --- /dev/null +++ b/rvs/conf/MI350P-600W/gst_single.conf @@ -0,0 +1,199 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# GST test - gst-96Tflops-8K12K4K-trig-tf32 +# +# Preconditions: +# Set device to all. If you need to run the rvs only on a subset of GPUs, please run rvs with -g +# option, collect the GPUs IDs (e.g.: GPU[ 5 - 50599] -> 50599 is the GPU ID) and then specify +# all the GPUs IDs separated by white space +# Set matrices sizes to 8192 * 12288 * 4096 +# Set matrix data type as fp32 real number +# Set compute type as tf32 (xf32) +# Set matrix data initialization method as trignometric float +# Set copy_matrix to false (the matrices will be copied to GPUs only once) +# Set target stress GFLOPS as 96 TFLOPS +# Set blas source (backend) as hipblaslt +# +# Expected result: +# The test on each GPU passes (TRUE) if the GPU achieves 96 TFLOPS or more +# within the test duration of 15 seconds after ramp-up duration of 5 seconds. +# Else test on the GPU fails (FALSE). + +actions: +- name: gst-8K8K16K-trig-bf16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 694000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-8K8K16K-trig-fp16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 665000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-8K8K16K-trig-fp8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 1513000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-4K4K32K-trig-fp4 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 2113200 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 32768 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp4_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + +- name: gst-Tflops-4K4K16K-trig-bf8 + module: gst + device: all + matrix_init: trig + target_stress: 0 + data_type: fp8_e5m2_r # BF8 (OCP E5M2) -> HIP_R_8F_E5M2 + out_data_type: bf16_r + compute_type: fp32_r + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 16384 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + duration: 15000 + hot_calls: 1000 + blas_source: hipblaslt + +- name: gst-Tflops-4K4K32K-trig-fp6 + module: gst + device: all + target_stress: 0 + matrix_init: trig + data_type: fp6_e3m2_r # MX FP6 E3M2 -> HIP_R_6F_E3M2 + out_data_type: fp16_r + compute_type: fp32_r + scale_a: block + scale_b: block + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 32768 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + duration: 15000 + blas_source: hipblaslt + +- name: gst-8K8K16K-rand-fp8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: rand + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt diff --git a/rvs/conf/MI350P-600W/iet_stress.conf b/rvs/conf/MI350P-600W/iet_stress.conf new file mode 100644 index 000000000..36b14fe7a --- /dev/null +++ b/rvs/conf/MI350P-600W/iet_stress.conf @@ -0,0 +1,61 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# IET stress test - MI350P long running power validation using typed GEMM via hipBLASLt +# +# Preconditions: +# Set device to all. If you need to run the rvs only on a subset of GPUs, please run rvs with -g +# option, collect the GPUs IDs (e.g.: GPU[ 5 - 50599] -> 50599 is the GPU ID) and then specify +# all the GPUs IDs separated by white space +# Set blas source (backend) as hipblaslt +# Set target power according to MI350P TDP +# +# Run test with: +# cd bin +# ./rvs -c conf/MI350P/iet_stress.conf +# +# Expected result: +# The test on each GPU passes (TRUE) if the GPU power reaches at least +# the target_power threshold, FALSE otherwise + +actions: +- name: iet-stress-600W-dgemm-true + device: all + module: iet + parallel: true + duration: 600000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + target_power: 580 + tolerance: 0.01 + matrix_size: 8192 + ops_type: dgemm + lda: 8192 + ldb: 8192 + ldc: 8192 + alpha: 1 + beta: 1 + matrix_init: hiprand diff --git a/rvs/conf/MI350P-600W/levels/rvs_level_1.conf b/rvs/conf/MI350P-600W/levels/rvs_level_1.conf new file mode 100644 index 000000000..14b8d6c62 --- /dev/null +++ b/rvs/conf/MI350P-600W/levels/rvs_level_1.conf @@ -0,0 +1,38 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: + +- name: metapackage-validation + device: all + module: rcqt + package: rocm-hip-sdk rocm-hip-libraries rocm-ml-libraries rocm-ml-sdk amdgpu-dkms rocm-language-runtime rocm-openmp-sdk rocm-utils rocm-opencl-sdk rocm-opencl-runtime rocm-hip-runtime rocm-developer-tools + +- name: packagelist-install-validation + device: all + module: rcqt + rpmpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-devel rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-devel hipcub-devel rocsolver rocsparse rocsolver-devel rocminfo hipfft-devel rocm-gdb rocm-dbgapi rocfft hipblas-devel rocthrust-devel openmp-extras-runtime openmp-extras-devel comgr rccl rocblas hipblas roctracer-devel hip-doc rocrand hsa-rocr hipfft hipsparse-devel rocsparse-devel rocrand-devel rocm-opencl hip-devel rocprim-devel hipsolver-devel rocfft-devel hsa-amd-aqlprofile hipify-clang miopen-hip-devel rocm-llvm hip-runtime-amd hip-samples rocalution-devel rccl-devel hipsolver rocprofiler-devel miopen-hip rocm-cmake hipsparse rocblas-devel rocm-opencl-devel + debpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-dev rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-dev hipcub-dev rocsolver rocsparse rocsolver-dev rocminfo hipfft-dev rocm-gdb rocm-dbgapi rocfft hipblas-dev rocthrust-dev openmp-extras-runtime openmp-extras-dev comgr rccl rocblas hipblas roctracer-dev hip-doc rocrand hsa-rocr hipfft hipsparse-dev rocsparse-dev rocrand-dev rocm-opencl hip-dev rocprim-dev hipsolver-dev rocfft-dev hsa-amd-aqlprofile hipify-clang miopen-hip-dev rocm-llvm hip-runtime-amd hip-samples rocalution-dev rccl-dev hipsolver rocprofiler-dev miopen-hip rocm-cmake hipsparse rocblas-dev rocm-opencl-dev + diff --git a/rvs/conf/MI350P-600W/levels/rvs_level_2.conf b/rvs/conf/MI350P-600W/levels/rvs_level_2.conf new file mode 100644 index 000000000..f34ed405f --- /dev/null +++ b/rvs/conf/MI350P-600W/levels/rvs_level_2.conf @@ -0,0 +1,79 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: pcie_capability + device: all + module: peqt + capability: + link_cap_max_speed: + link_cap_max_width: + link_stat_cur_speed: + link_stat_neg_width: + slot_pwr_limit_value: + slot_physical_num: + deviceid: + vendor_id: + kernel_driver: + +- name: hbm_basic + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 865075200 + test_type: 2 + mibibytes: false + o/p_csv: false + read: true + write: true + subtest: 0 + dwords_per_lane: 4 + chunks_per_block: 1 + tb_size: 512 + +- name: xgmi_interface + device: all + module: pbqt + log_interval: 5000 + duration: 20000 + peers: all + test_bandwidth: true + bidirectional: false + parallel: true + block_size: 67108864 + device_id: all + +- name: pcie_interface + device: all + module: pebb + duration: 20000 + device_to_host: false + host_to_device: true + parallel: true + block_size: 67108864 + link_type: 2 + diff --git a/rvs/conf/MI350P-600W/levels/rvs_level_3.conf b/rvs/conf/MI350P-600W/levels/rvs_level_3.conf new file mode 100644 index 000000000..6a8a400ac --- /dev/null +++ b/rvs/conf/MI350P-600W/levels/rvs_level_3.conf @@ -0,0 +1,252 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# MI350P-600W test level 3 - basic performance +# Calibrated GST target_stress values (fp8/fp16/bf16) sourced from +# MI350P-600W/gst_single.conf. Uncalibrated dtypes/shapes use 0. + +actions: +- name: hbm_mid + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 865075200 + test_type: 2 + mibibytes: false + o/p_csv: false + read: true + write: true + copy: true + mul: true + add: true + dwords_per_lane: 4 + chunks_per_block: 1 + tb_size: 512 + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: compute-fp4-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp4_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e3m2_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-bf6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e2m3_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 1513000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 665000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 694000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + diff --git a/rvs/conf/MI350P-600W/levels/rvs_level_4.conf b/rvs/conf/MI350P-600W/levels/rvs_level_4.conf new file mode 100644 index 000000000..3c40f7b6a --- /dev/null +++ b/rvs/conf/MI350P-600W/levels/rvs_level_4.conf @@ -0,0 +1,335 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# MI350P-600W test level 4 - extended performance +# Calibrated GST target_stress (fp8/fp16/bf16) sourced from MI350P-600W/gst_single.conf. +# IET target_power = 580 W (per MI350P-600W TDP, matching iet_stress.conf). + +actions: +- name: hbm_full + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 865075200 + test_type: 2 + mibibytes: false + o/p_csv: false + read: true + write: true + copy: true + mul: true + add: true + triad: true + dot: true + dwords_per_lane: 4 + chunks_per_block: 1 + tb_size: 512 + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: memtest + device: all + module: mem + parallel: true + count: 1 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + +- name: compute-fp4-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp4_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e3m2_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-bf6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e2m3_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 1513000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 665000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 694000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-fp32-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: trig + data_type: fp32_r + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-fp64-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp64_r + compute_type: fp64_r + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: power-stress + device: all + module: iet + parallel: true + duration: 60000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + target_power: 580 + tolerance: 0.01 + matrix_size: 8192 + ops_type: dgemm + lda: 8192 + ldb: 8192 + ldc: 8192 + alpha: 1 + beta: 1 + matrix_init: hiprand + diff --git a/rvs/conf/MI350P-600W/levels/rvs_level_5.conf b/rvs/conf/MI350P-600W/levels/rvs_level_5.conf new file mode 100644 index 000000000..4fe4ad4a4 --- /dev/null +++ b/rvs/conf/MI350P-600W/levels/rvs_level_5.conf @@ -0,0 +1,335 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# MI350P-600W test level 5 - full stress test (long duration) +# Calibrated GST target_stress (fp8/fp16/bf16) sourced from MI350P-600W/gst_single.conf. +# IET target_power = 580 W (per MI350P-600W TDP). + +actions: +- name: hbm_full + device: all + module: babel + parallel: true + count: 1 + num_iter: 10000 + array_size: 865075200 + test_type: 2 + mibibytes: false + o/p_csv: false + read: true + write: true + copy: true + mul: true + add: true + triad: true + dot: true + dwords_per_lane: 4 + chunks_per_block: 1 + tb_size: 512 + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 900000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: memtest + device: all + module: mem + parallel: true + count: 20 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + +- name: compute-fp4-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp4_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e3m2_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-bf6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e2m3_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 10000 + copy_matrix: false + target_stress: 1513000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 10000 + copy_matrix: false + target_stress: 665000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 10000 + copy_matrix: false + target_stress: 694000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-fp32-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: trig + data_type: fp32_r + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-fp64-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp64_r + compute_type: fp64_r + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: power-stress + device: all + module: iet + parallel: true + duration: 3600000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + target_power: 580 + tolerance: 0.01 + matrix_size: 8192 + ops_type: dgemm + lda: 8192 + ldb: 8192 + ldc: 8192 + alpha: 1 + beta: 1 + matrix_init: hiprand + diff --git a/rvs/conf/MI350P-600W/pbqt_single.conf b/rvs/conf/MI350P-600W/pbqt_single.conf new file mode 100644 index 000000000..e11290e04 --- /dev/null +++ b/rvs/conf/MI350P-600W/pbqt_single.conf @@ -0,0 +1,118 @@ +# ################################################################################ +# # +# # Copyright (c) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### +# +# PBQT peer bandwidth tests for AMD Instinct MI350P (PCIe-attached). +# MI350P GPUs are connected via PCIe (typical hop count 2), not XGMI. Do not use +# MI350X configs: TransferBench all-to-all defaults to a2a_direct=1, which only +# includes hop-count-1 (direct XGMI) pairs and yields no valid transfers on MI350P. +# +# Usage: +# ./rvs -c conf/MI350P-600W/pbqt_single.conf +# ./rvs -m pbqt + +actions: +# Peer-to-Peer/Device-to-Device/GPU-to-GPU unidirectional bandwidth test +# using RVS native method via SDMA +- name: pcie_d2d_unidir_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: false + parallel: false + block_size: 1073741824 + device_id: all + +# Peer-to-Peer/Device-to-Device/GPU-to-GPU bidirectional bandwidth test +# using RVS native method via SDMA +- name: pcie_d2d_bidir_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +# Peer-to-Peer/Device-to-Device/GPU-to-GPU unidirectional bandwidth test +# using TransferBench via SDMA +- name: pcie_d2d_unidir_tb_dma + device: all + module: pbqt + duration: 0 + peers: all + test_bandwidth: true + bidirectional: false + parallel: false + block_size: 1073741824 + device_id: all + transfer_method: transferbench + executor: dma + subexecutor: 1 + warm_calls: 2 + hot_calls: 5 + +# Peer-to-Peer/Device-to-Device/GPU-to-GPU unidirectional bandwidth test +# using TransferBench via GFX +- name: pcie_d2d_unidir_tb_gfx + device: all + module: pbqt + duration: 0 + peers: all + test_bandwidth: true + bidirectional: false + parallel: false + block_size: 1073741824 + device_id: all + transfer_method: transferbench + executor: gfx + subexecutor: 16 + warm_calls: 2 + hot_calls: 5 + +# All-to-All bandwidth test using TransferBench via GFX +# a2a_direct: 0 — include PCIe-connected pairs (hop count > 1); default 1 is XGMI-only +- name: pcie_d2d_unidir_tb_a2a + device: all + module: pbqt + peers: all + test_bandwidth: true + bidirectional: false + parallel: false + block_size: 268435456 + device_id: all + transfer_method: transferbench + transferbench_test: alltoall + executor: gfx + subexecutor: 8 + warm_calls: 3 + hot_calls: 10 + gfx_unroll: 2 + a2a_direct: 0 diff --git a/rvs/conf/MI350P-600W/pebb_single.conf b/rvs/conf/MI350P-600W/pebb_single.conf new file mode 100644 index 000000000..06a8ecb58 --- /dev/null +++ b/rvs/conf/MI350P-600W/pebb_single.conf @@ -0,0 +1,127 @@ +# ################################################################################ +# # +# # Copyright (c) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### +# +# PEBB bandwidth tests for AMD Instinct MI350P (PCIe-attached). +# Host/device transfers use link_type 2 (PCIe). Do not use MI350X configs on +# MI350P; GPU-to-GPU topology differs (PCIe hop count 2 vs XGMI on MI350X). +# +# Usage: +# ./rvs -c conf/MI350P-600W/pebb_single.conf +# ./rvs -m pebb + +actions: +# Host-to-Device/CPU-to-GPU bandwidth test using RVS native method via SDMA +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +# Device-to-Host/GPU-to-CPU bandwidth test using RVS native method via SDMA +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +# Host-to-Device/CPU-to-GPU bandwidth test using TransferBench via SDMA +- name: pcie_h2d_bandwidth_tb_dma + device: all + module: pebb + duration: 0 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + transfer_method: transferbench + executor: dma + subexecutor: 1 + warm_calls: 3 + hot_calls: 10 + source_memory: cpu + destination_memory: gpu + +# Device-to-Host/GPU-to-CPU bandwidth test using TransferBench via SDMA +- name: pcie_d2h_bandwidth_tb_dma + device: all + module: pebb + duration: 0 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + transfer_method: transferbench + executor: dma + subexecutor: 1 + warm_calls: 3 + hot_calls: 10 + source_memory: gpu + destination_memory: cpu + +# Host-to-Device/CPU-to-GPU bandwidth test using TransferBench via GFX +- name: pcie_h2d_bandwidth_tb_gfx + device: all + module: pebb + duration: 0 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + transfer_method: transferbench + executor: gfx + subexecutor: 8 + warm_calls: 3 + hot_calls: 10 + source_memory: cpu + destination_memory: "null" + +# Device-to-Host/GPU-to-CPU bandwidth test using TransferBench via GFX +- name: pcie_d2h_bandwidth_tb_gfx + device: all + module: pebb + duration: 0 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + transfer_method: transferbench + executor: gfx + subexecutor: 8 + warm_calls: 3 + hot_calls: 10 + source_memory: "null" + destination_memory: cpu diff --git a/rvs/conf/MI350X/NPS2/CPX/gst_single.conf b/rvs/conf/MI350X/NPS2/CPX/gst_single.conf new file mode 100644 index 000000000..437f63b6d --- /dev/null +++ b/rvs/conf/MI350X/NPS2/CPX/gst_single.conf @@ -0,0 +1,393 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# GST test - gst-Tflops-2K2K2K-trig-fp4 +# +# Preconditions: +# Set device to all. If you need to run the rvs only on a subset of GPUs, please run rvs with -g +# option, collect the GPUs IDs (e.g.: GPU[ 5 - 50599] -> 50599 is the GPU ID) and then specify +# all the GPUs IDs separated by white space +# Set matrices sizes to 2048 * 2048 * 2048 +# Set matrix data type as fp4 real number +# Set matrix data initialization method as trignometric float +# Set copy_matrix to false (the matrices will be copied to GPUs only once) +# Set target stress GFLOPS as XXXX +# +# Expected result: +# The test on each GPU passes (TRUE) if the GPU achieves XXXX TFLOPS or more +# within the test duration of 15 seconds after ramp-up duration of 5 seconds. +# Else test on the GPU fails (FALSE). + +actions: +- name: gst-Tflops-2K2K2K-trig-fp4 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp4_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + +- name: gst-Tflops-2K2K2K-trig-fp6 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e3m2_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + +- name: gst-Tflops-2K2K2K-trig-bf6 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e2m3_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-trig-fp8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-rand-fp8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: rand + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-trig-bf8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-rand-bf8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: rand + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-trig-fp16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-rand-fp16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: rand + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-trig-bf16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-rand-bf16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: rand + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-3K-trig-fp32 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: trig + data_type: fp32_r + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + +- name: gst-Tflops-3K-rand-fp32 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: rand + data_type: fp32_r + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + +- name: gst-Tflops-8K-trig-fp64 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp64_r + compute_type: fp64_r + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + +- name: gst-Tflops-8K-rand-fp64 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: rand + data_type: fp64_r + compute_type: fp64_r + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + diff --git a/rvs/conf/MI350X/NPS2/DPX/gst_single.conf b/rvs/conf/MI350X/NPS2/DPX/gst_single.conf new file mode 100644 index 000000000..437f63b6d --- /dev/null +++ b/rvs/conf/MI350X/NPS2/DPX/gst_single.conf @@ -0,0 +1,393 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# GST test - gst-Tflops-2K2K2K-trig-fp4 +# +# Preconditions: +# Set device to all. If you need to run the rvs only on a subset of GPUs, please run rvs with -g +# option, collect the GPUs IDs (e.g.: GPU[ 5 - 50599] -> 50599 is the GPU ID) and then specify +# all the GPUs IDs separated by white space +# Set matrices sizes to 2048 * 2048 * 2048 +# Set matrix data type as fp4 real number +# Set matrix data initialization method as trignometric float +# Set copy_matrix to false (the matrices will be copied to GPUs only once) +# Set target stress GFLOPS as XXXX +# +# Expected result: +# The test on each GPU passes (TRUE) if the GPU achieves XXXX TFLOPS or more +# within the test duration of 15 seconds after ramp-up duration of 5 seconds. +# Else test on the GPU fails (FALSE). + +actions: +- name: gst-Tflops-2K2K2K-trig-fp4 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp4_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + +- name: gst-Tflops-2K2K2K-trig-fp6 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e3m2_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + +- name: gst-Tflops-2K2K2K-trig-bf6 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e2m3_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-trig-fp8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-rand-fp8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: rand + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-trig-bf8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-rand-bf8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: rand + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-trig-fp16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-rand-fp16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: rand + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-trig-bf16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-rand-bf16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: rand + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-3K-trig-fp32 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: trig + data_type: fp32_r + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + +- name: gst-Tflops-3K-rand-fp32 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: rand + data_type: fp32_r + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + +- name: gst-Tflops-8K-trig-fp64 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp64_r + compute_type: fp64_r + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + +- name: gst-Tflops-8K-rand-fp64 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: rand + data_type: fp64_r + compute_type: fp64_r + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + diff --git a/rvs/conf/MI350X/NPS2/QPX/gst_single.conf b/rvs/conf/MI350X/NPS2/QPX/gst_single.conf new file mode 100644 index 000000000..80ddded1a --- /dev/null +++ b/rvs/conf/MI350X/NPS2/QPX/gst_single.conf @@ -0,0 +1,393 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# GST test - gst-Tflops-2K2K2K-trig-fp4 +# +# Preconditions: +# Set device to all. If you need to run the rvs only on a subset of GPUs, please run rvs with -g +# option, collect the GPUs IDs (e.g.: GPU[ 5 - 50599] -> 50599 is the GPU ID) and then specify +# all the GPUs IDs separated by white space +# Set matrices sizes to 2048 * 2048 * 2048 +# Set matrix data type as fp4 real number +# Set matrix data initialization method as trignometric float +# Set copy_matrix to false (the matrices will be copied to GPUs only once) +# Set target stress GFLOPS as XXXX +# +# Expected result: +# The test on each GPU passes (TRUE) if the GPU achieves XXXX TFLOPS or more +# within the test duration of 15 seconds after ramp-up duration of 5 seconds. +# Else test on the GPU fails (FALSE). + +actions: +- name: gst-Tflops-2K2K2K-trig-fp4 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp4_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + +- name: gst-Tflops-2K2K2K-trig-fp6 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e3m2_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + +- name: gst-Tflops-2K2K2K-trig-bf6 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e2m3_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-trig-fp8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-rand-fp8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: rand + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-trig-bf8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-rand-bf8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: rand + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-trig-fp16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-rand-fp16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: rand + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-trig-bf16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-rand-bf16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: rand + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-3K-trig-fp32 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: trig + data_type: fp32_r + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + +- name: gst-Tflops-3K-rand-fp32 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: rand + data_type: fp32_r + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + +- name: gst-Tflops-8K-trig-fp64 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp64_r + compute_type: fp64_r + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + +- name: gst-Tflops-8K-rand-fp64 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: rand + data_type: fp64_r + compute_type: fp64_r + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + diff --git a/rvs/conf/MI350X/babel.conf b/rvs/conf/MI350X/babel.conf index 6a5e1bbc0..5d918449b 100644 --- a/rvs/conf/MI350X/babel.conf +++ b/rvs/conf/MI350X/babel.conf @@ -1,6 +1,6 @@ # ################################################################################ # # -# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # Copyright (c) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. # # # # MIT LICENSE: # # Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -40,13 +40,19 @@ actions: module: babel # Name of the module parallel: false # Parallel true or false count: 1 # Number of times you want to repeat the test from the begin ( A clean start every time) - num_iter: 2000 # Number of iterations, this many kernels are launched simultaneosuly and stresses the system + num_iter: 3000 # Number of iterations, this many kernels are launched simultaneosuly and stresses the system + duration: 0 # Duration in milliseconds. When > 0, runs for this time instead of num_iter. 0 = use num_iter array_size: 865075200 # Array size the test operates, this is 825 MiB test_type: 2 # type of test, 1: Float, 2: Double, 3: Triad float, 4: Triad double mibibytes: false # mibibytes (MiB) or megabytes (MB), true for MiB o/p_csv: false # o/p as csv file - rwtest: 2 # 1: read 2: read+write - subtest: 5 # 1: copy 2: copy+mul 3: copy+mul+add 4: copy+mul+add+traid 5: copy+mul+add+traid+dot + read: true # read operation + write: true # write operation + copy: true # copy operation + mul: true # mul operation + add: true # add operation + triad: true # triad operation + dot: true # dot operation dwords_per_lane: 4 # Number of dwords per lane chunks_per_block: 1 # Number of chunks per block tb_size: 512 # Thread block size diff --git a/rvs/conf/MI350X/gst_single.conf b/rvs/conf/MI350X/gst_single.conf index 912c86e85..ae634a1fe 100644 --- a/rvs/conf/MI350X/gst_single.conf +++ b/rvs/conf/MI350X/gst_single.conf @@ -357,6 +357,7 @@ actions: matrix_size_c: 8192 matrix_init: trig data_type: fp64_r + compute_type: fp64_r lda: 8192 ldb: 8192 ldc: 8192 @@ -380,6 +381,7 @@ actions: matrix_size_c: 8192 matrix_init: rand data_type: fp64_r + compute_type: fp64_r lda: 8192 ldb: 8192 ldc: 8192 diff --git a/rvs/conf/MI350X/levels/rvs_level_1.conf b/rvs/conf/MI350X/levels/rvs_level_1.conf new file mode 100644 index 000000000..0d3512230 --- /dev/null +++ b/rvs/conf/MI350X/levels/rvs_level_1.conf @@ -0,0 +1,38 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: + +- name: metapackage-validation + device: all + module: rcqt + package: rocm-hip-sdk rocm-hip-libraries rocm-ml-libraries rocm-ml-sdk amdgpu-dkms rocm-language-runtime rocm-openmp-sdk rocm-utils rocm-opencl-sdk rocm-opencl-runtime rocm-hip-runtime rocm-developer-tools + +- name: packagelist-install-validation + device: all + module: rcqt + rpmpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-devel rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-devel hipcub-devel rocsolver rocsparse rocsolver-devel rocminfo hipfft-devel rocm-gdb rocm-dbgapi rocfft hipblas-devel rocthrust-devel openmp-extras-runtime openmp-extras-devel comgr rccl rocblas hipblas roctracer-devel hip-doc rocrand hsa-rocr hipfft hipsparse-devel rocsparse-devel rocrand-devel rocm-opencl hip-devel rocprim-devel hipsolver-devel rocfft-devel hsa-amd-aqlprofile hipify-clang miopen-hip-devel rocm-llvm hip-runtime-amd hip-samples rocalution-devel rccl-devel hipsolver rocprofiler-devel miopen-hip rocm-cmake hipsparse rocblas-devel rocm-opencl-devel + debpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-dev rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-dev hipcub-dev rocsolver rocsparse rocsolver-dev rocminfo hipfft-dev rocm-gdb rocm-dbgapi rocfft hipblas-dev rocthrust-dev openmp-extras-runtime openmp-extras-dev comgr rccl rocblas hipblas roctracer-dev hip-doc rocrand hsa-rocr hipfft hipsparse-dev rocsparse-dev rocrand-dev rocm-opencl hip-dev rocprim-dev hipsolver-dev rocfft-dev hsa-amd-aqlprofile hipify-clang miopen-hip-dev rocm-llvm hip-runtime-amd hip-samples rocalution-dev rccl-dev hipsolver rocprofiler-dev miopen-hip rocm-cmake hipsparse rocblas-dev rocm-opencl-dev + diff --git a/rvs/conf/MI350X/levels/rvs_level_2.conf b/rvs/conf/MI350X/levels/rvs_level_2.conf new file mode 100644 index 000000000..c6bde5336 --- /dev/null +++ b/rvs/conf/MI350X/levels/rvs_level_2.conf @@ -0,0 +1,79 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: pcie_capability + device: all + module: peqt + capability: + link_cap_max_speed: + link_cap_max_width: + link_stat_cur_speed: + link_stat_neg_width: + slot_pwr_limit_value: + slot_physical_num: + deviceid: + vendor_id: + kernel_driver: + +- name: hbm_basic + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 865075200 + test_type: 2 + mibibytes: false + o/p_csv: false + read: true + write: true + subtest: 0 + dwords_per_lane: 4 + chunks_per_block: 1 + tb_size: 512 + +- name: xgmi_interface + device: all + module: pbqt + log_interval: 5000 + duration: 20000 + peers: all + test_bandwidth: true + bidirectional: false + parallel: true + block_size: 67108864 + device_id: all + +- name: pcie_interface + device: all + module: pebb + duration: 20000 + device_to_host: false + host_to_device: true + parallel: true + block_size: 67108864 + link_type: 2 + diff --git a/rvs/conf/MI350X/levels/rvs_level_3.conf b/rvs/conf/MI350X/levels/rvs_level_3.conf new file mode 100644 index 000000000..9106e7dbb --- /dev/null +++ b/rvs/conf/MI350X/levels/rvs_level_3.conf @@ -0,0 +1,248 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: hbm_mid + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 865075200 + test_type: 2 + mibibytes: false + o/p_csv: false + read: true + write: true + copy: true + mul: true + add: true + dwords_per_lane: 4 + chunks_per_block: 1 + tb_size: 512 + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: compute-fp4-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp4_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e3m2_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-bf6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e2m3_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 2061000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 1854000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 946000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 956000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + diff --git a/rvs/conf/MI350X/levels/rvs_level_4.conf b/rvs/conf/MI350X/levels/rvs_level_4.conf new file mode 100644 index 000000000..ae9497516 --- /dev/null +++ b/rvs/conf/MI350X/levels/rvs_level_4.conf @@ -0,0 +1,331 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: hbm_full + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 865075200 + test_type: 2 + mibibytes: false + o/p_csv: false + read: true + write: true + copy: true + mul: true + add: true + triad: true + dot: true + dwords_per_lane: 4 + chunks_per_block: 1 + tb_size: 512 + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: memtest + device: all + module: mem + parallel: true + count: 1 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + +- name: compute-fp4-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp4_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e3m2_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-bf6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e2m3_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 2061000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 1854000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 946000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 956000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-fp32-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: trig + data_type: fp32_r + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-fp64-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp64_r + compute_type: fp64_r + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: power-stress + device: all + module: iet + parallel: true + duration: 60000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + target_power: 1000 + tolerance: 0.01 + matrix_size: 28000 + ops_type: dgemm + lda: 28000 + ldb: 28000 + ldc: 28000 + alpha: 1 + beta: 1 + matrix_init: hiprand + diff --git a/rvs/conf/MI350X/levels/rvs_level_5.conf b/rvs/conf/MI350X/levels/rvs_level_5.conf new file mode 100644 index 000000000..190c5c5ab --- /dev/null +++ b/rvs/conf/MI350X/levels/rvs_level_5.conf @@ -0,0 +1,331 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: hbm_full + device: all + module: babel + parallel: true + count: 1 + num_iter: 10000 + array_size: 865075200 + test_type: 2 + mibibytes: false + o/p_csv: false + read: true + write: true + copy: true + mul: true + add: true + triad: true + dot: true + dwords_per_lane: 4 + chunks_per_block: 1 + tb_size: 512 + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 900000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: memtest + device: all + module: mem + parallel: true + count: 20 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + +- name: compute-fp4-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp4_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e3m2_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-bf6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e2m3_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 10000 + copy_matrix: false + target_stress: 2061000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 10000 + copy_matrix: false + target_stress: 1854000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 10000 + copy_matrix: false + target_stress: 946000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 10000 + copy_matrix: false + target_stress: 956000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-fp32-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: trig + data_type: fp32_r + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-fp64-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp64_r + compute_type: fp64_r + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: power-stress + device: all + module: iet + parallel: true + duration: 3600000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + target_power: 1000 + tolerance: 0.01 + matrix_size: 28000 + ops_type: dgemm + lda: 28000 + ldb: 28000 + ldc: 28000 + alpha: 1 + beta: 1 + matrix_init: hiprand + diff --git a/rvs/conf/MI350X/pbqt_single.conf b/rvs/conf/MI350X/pbqt_single.conf new file mode 100644 index 000000000..d950c4301 --- /dev/null +++ b/rvs/conf/MI350X/pbqt_single.conf @@ -0,0 +1,108 @@ +# ################################################################################ +# # +# # Copyright (c) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +# Peer-to-Peer/Device-to-Device/GPU-to-GPU unidirectional bandwidth test +# using RVS native method via SDMA +- name: xgmi_d2d_unidir_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: false + parallel: false + block_size: 1073741824 + device_id: all + +# Peer-to-Peer/Device-to-Device/GPU-to-GPU bidirectional bandwidth test +# using RVS native method via SDMA +- name: xgmi_d2d_bidir_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +# Peer-to-Peer/Device-to-Device/GPU-to-GPU unidirectional bandwidth test +# using TransferBench via SDMA +- name: xgmi_d2d_unidir_tb_dma + device: all + module: pbqt + duration: 0 + peers: all + test_bandwidth: true + bidirectional: false + parallel: false + block_size: 1073741824 + device_id: all + transfer_method: transferbench + executor: dma + subexecutor: 1 + warm_calls: 2 + hot_calls: 5 + +# Peer-to-Peer/Device-to-Device/GPU-to-GPU unidirectional bandwidth test +# using TransferBench via GFX +- name: xgmi_d2d_unidir_tb_gfx + device: all + module: pbqt + duration: 0 + peers: all + test_bandwidth: true + bidirectional: false + parallel: false + block_size: 1073741824 + device_id: all + transfer_method: transferbench + executor: gfx + subexecutor: 16 + warm_calls: 2 + hot_calls: 5 + +# All-to-All bandwidth test using TransferBench via GFX +- name: xgmi_d2d_unidir_tb_a2a + device: all + module: pbqt + peers: all + test_bandwidth: true + bidirectional: false + parallel: false + block_size: 268435456 + device_id: all + transfer_method: transferbench + transferbench_test: alltoall + executor: gfx + subexecutor: 8 + warm_calls: 3 + hot_calls: 10 + gfx_unroll: 2 + diff --git a/rvs/conf/MI350X/pebb_single.conf b/rvs/conf/MI350X/pebb_single.conf new file mode 100644 index 000000000..a20fc9789 --- /dev/null +++ b/rvs/conf/MI350X/pebb_single.conf @@ -0,0 +1,120 @@ +# ################################################################################ +# # +# # Copyright (c) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +# Host-to-Device/CPU-to-GPU bandwidth test using RVS native method via SDMA +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +# Device-to-Host/GPU-to-CPU bandwidth test using RVS native method via SDMA +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +# Host-to-Device/CPU-to-GPU bandwidth test using TransferBench via SDMA +- name: pcie_h2d_bandwidth_tb_dma + device: all + module: pebb + duration: 0 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + transfer_method: transferbench + executor: dma + subexecutor: 1 + warm_calls: 3 + hot_calls: 10 + source_memory: cpu + destination_memory: gpu + +# Device-to-Host/GPU-to-CPU bandwidth test using TransferBench via SDMA +- name: pcie_d2h_bandwidth_tb_dma + device: all + module: pebb + duration: 0 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + transfer_method: transferbench + executor: dma + subexecutor: 1 + warm_calls: 3 + hot_calls: 10 + source_memory: gpu + destination_memory: cpu + +# Host-to-Device/CPU-to-GPU bandwidth test using TransferBench via GFX +- name: pcie_h2d_bandwidth_tb_gfx + device: all + module: pebb + duration: 0 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + transfer_method: transferbench + executor: gfx + subexecutor: 8 + warm_calls: 3 + hot_calls: 10 + source_memory: cpu + destination_memory: "null" + +# Device-to-Host/GPU-to-CPU bandwidth test using TransferBench via GFX +- name: pcie_d2h_bandwidth_tb_gfx + device: all + module: pebb + duration: 0 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + transfer_method: transferbench + executor: gfx + subexecutor: 8 + warm_calls: 3 + hot_calls: 10 + source_memory: "null" + destination_memory: cpu + diff --git a/rvs/conf/MI355X/NPS2/CPX/gst_single.conf b/rvs/conf/MI355X/NPS2/CPX/gst_single.conf new file mode 100644 index 000000000..437f63b6d --- /dev/null +++ b/rvs/conf/MI355X/NPS2/CPX/gst_single.conf @@ -0,0 +1,393 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# GST test - gst-Tflops-2K2K2K-trig-fp4 +# +# Preconditions: +# Set device to all. If you need to run the rvs only on a subset of GPUs, please run rvs with -g +# option, collect the GPUs IDs (e.g.: GPU[ 5 - 50599] -> 50599 is the GPU ID) and then specify +# all the GPUs IDs separated by white space +# Set matrices sizes to 2048 * 2048 * 2048 +# Set matrix data type as fp4 real number +# Set matrix data initialization method as trignometric float +# Set copy_matrix to false (the matrices will be copied to GPUs only once) +# Set target stress GFLOPS as XXXX +# +# Expected result: +# The test on each GPU passes (TRUE) if the GPU achieves XXXX TFLOPS or more +# within the test duration of 15 seconds after ramp-up duration of 5 seconds. +# Else test on the GPU fails (FALSE). + +actions: +- name: gst-Tflops-2K2K2K-trig-fp4 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp4_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + +- name: gst-Tflops-2K2K2K-trig-fp6 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e3m2_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + +- name: gst-Tflops-2K2K2K-trig-bf6 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e2m3_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-trig-fp8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-rand-fp8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: rand + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-trig-bf8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-rand-bf8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: rand + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-trig-fp16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-rand-fp16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: rand + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-trig-bf16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-rand-bf16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: rand + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-3K-trig-fp32 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: trig + data_type: fp32_r + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + +- name: gst-Tflops-3K-rand-fp32 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: rand + data_type: fp32_r + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + +- name: gst-Tflops-8K-trig-fp64 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp64_r + compute_type: fp64_r + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + +- name: gst-Tflops-8K-rand-fp64 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: rand + data_type: fp64_r + compute_type: fp64_r + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + diff --git a/rvs/conf/MI355X/NPS2/DPX/gst_single.conf b/rvs/conf/MI355X/NPS2/DPX/gst_single.conf new file mode 100644 index 000000000..437f63b6d --- /dev/null +++ b/rvs/conf/MI355X/NPS2/DPX/gst_single.conf @@ -0,0 +1,393 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# GST test - gst-Tflops-2K2K2K-trig-fp4 +# +# Preconditions: +# Set device to all. If you need to run the rvs only on a subset of GPUs, please run rvs with -g +# option, collect the GPUs IDs (e.g.: GPU[ 5 - 50599] -> 50599 is the GPU ID) and then specify +# all the GPUs IDs separated by white space +# Set matrices sizes to 2048 * 2048 * 2048 +# Set matrix data type as fp4 real number +# Set matrix data initialization method as trignometric float +# Set copy_matrix to false (the matrices will be copied to GPUs only once) +# Set target stress GFLOPS as XXXX +# +# Expected result: +# The test on each GPU passes (TRUE) if the GPU achieves XXXX TFLOPS or more +# within the test duration of 15 seconds after ramp-up duration of 5 seconds. +# Else test on the GPU fails (FALSE). + +actions: +- name: gst-Tflops-2K2K2K-trig-fp4 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp4_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + +- name: gst-Tflops-2K2K2K-trig-fp6 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e3m2_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + +- name: gst-Tflops-2K2K2K-trig-bf6 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e2m3_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-trig-fp8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-rand-fp8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: rand + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-trig-bf8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-rand-bf8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: rand + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-trig-fp16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-rand-fp16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: rand + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-trig-bf16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-rand-bf16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: rand + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-3K-trig-fp32 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: trig + data_type: fp32_r + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + +- name: gst-Tflops-3K-rand-fp32 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: rand + data_type: fp32_r + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + +- name: gst-Tflops-8K-trig-fp64 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp64_r + compute_type: fp64_r + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + +- name: gst-Tflops-8K-rand-fp64 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: rand + data_type: fp64_r + compute_type: fp64_r + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + diff --git a/rvs/conf/MI355X/NPS2/QPX/gst_single.conf b/rvs/conf/MI355X/NPS2/QPX/gst_single.conf new file mode 100644 index 000000000..80ddded1a --- /dev/null +++ b/rvs/conf/MI355X/NPS2/QPX/gst_single.conf @@ -0,0 +1,393 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# GST test - gst-Tflops-2K2K2K-trig-fp4 +# +# Preconditions: +# Set device to all. If you need to run the rvs only on a subset of GPUs, please run rvs with -g +# option, collect the GPUs IDs (e.g.: GPU[ 5 - 50599] -> 50599 is the GPU ID) and then specify +# all the GPUs IDs separated by white space +# Set matrices sizes to 2048 * 2048 * 2048 +# Set matrix data type as fp4 real number +# Set matrix data initialization method as trignometric float +# Set copy_matrix to false (the matrices will be copied to GPUs only once) +# Set target stress GFLOPS as XXXX +# +# Expected result: +# The test on each GPU passes (TRUE) if the GPU achieves XXXX TFLOPS or more +# within the test duration of 15 seconds after ramp-up duration of 5 seconds. +# Else test on the GPU fails (FALSE). + +actions: +- name: gst-Tflops-2K2K2K-trig-fp4 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp4_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + +- name: gst-Tflops-2K2K2K-trig-fp6 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e3m2_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + +- name: gst-Tflops-2K2K2K-trig-bf6 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e2m3_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-trig-fp8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-rand-fp8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: rand + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-trig-bf8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-rand-bf8 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: rand + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-trig-fp16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-rand-fp16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: rand + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-trig-bf16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-8K8K16K-rand-bf16 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: rand + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + +- name: gst-Tflops-3K-trig-fp32 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: trig + data_type: fp32_r + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + +- name: gst-Tflops-3K-rand-fp32 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: rand + data_type: fp32_r + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + +- name: gst-Tflops-8K-trig-fp64 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp64_r + compute_type: fp64_r + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + +- name: gst-Tflops-8K-rand-fp64 + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: rand + data_type: fp64_r + compute_type: fp64_r + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + diff --git a/rvs/conf/MI355X/babel.conf b/rvs/conf/MI355X/babel.conf index 6a5e1bbc0..1da11e2e8 100644 --- a/rvs/conf/MI355X/babel.conf +++ b/rvs/conf/MI355X/babel.conf @@ -1,6 +1,6 @@ # ################################################################################ # # -# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # Copyright (c) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. # # # # MIT LICENSE: # # Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -40,13 +40,19 @@ actions: module: babel # Name of the module parallel: false # Parallel true or false count: 1 # Number of times you want to repeat the test from the begin ( A clean start every time) - num_iter: 2000 # Number of iterations, this many kernels are launched simultaneosuly and stresses the system + num_iter: 3000 # Number of iterations, this many kernels are launched simultaneosuly and stresses the system + duration: 0 # Duration in milliseconds. When > 0, runs for this time instead of num_iter. 0 = use num_iter array_size: 865075200 # Array size the test operates, this is 825 MiB test_type: 2 # type of test, 1: Float, 2: Double, 3: Triad float, 4: Triad double mibibytes: false # mibibytes (MiB) or megabytes (MB), true for MiB o/p_csv: false # o/p as csv file - rwtest: 2 # 1: read 2: read+write - subtest: 5 # 1: copy 2: copy+mul 3: copy+mul+add 4: copy+mul+add+traid 5: copy+mul+add+traid+dot + read: true # read operation + write: true # write operation + copy: true # copy operation + mul: true # mul operation + add: true # add operation + dot: true # dot operation + triad: true # triad operation dwords_per_lane: 4 # Number of dwords per lane chunks_per_block: 1 # Number of chunks per block tb_size: 512 # Thread block size diff --git a/rvs/conf/MI355X/gst_single.conf b/rvs/conf/MI355X/gst_single.conf index 912c86e85..ae634a1fe 100644 --- a/rvs/conf/MI355X/gst_single.conf +++ b/rvs/conf/MI355X/gst_single.conf @@ -357,6 +357,7 @@ actions: matrix_size_c: 8192 matrix_init: trig data_type: fp64_r + compute_type: fp64_r lda: 8192 ldb: 8192 ldc: 8192 @@ -380,6 +381,7 @@ actions: matrix_size_c: 8192 matrix_init: rand data_type: fp64_r + compute_type: fp64_r lda: 8192 ldb: 8192 ldc: 8192 diff --git a/rvs/conf/MI355X/levels/rvs_level_1.conf b/rvs/conf/MI355X/levels/rvs_level_1.conf new file mode 100644 index 000000000..0d3512230 --- /dev/null +++ b/rvs/conf/MI355X/levels/rvs_level_1.conf @@ -0,0 +1,38 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: + +- name: metapackage-validation + device: all + module: rcqt + package: rocm-hip-sdk rocm-hip-libraries rocm-ml-libraries rocm-ml-sdk amdgpu-dkms rocm-language-runtime rocm-openmp-sdk rocm-utils rocm-opencl-sdk rocm-opencl-runtime rocm-hip-runtime rocm-developer-tools + +- name: packagelist-install-validation + device: all + module: rcqt + rpmpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-devel rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-devel hipcub-devel rocsolver rocsparse rocsolver-devel rocminfo hipfft-devel rocm-gdb rocm-dbgapi rocfft hipblas-devel rocthrust-devel openmp-extras-runtime openmp-extras-devel comgr rccl rocblas hipblas roctracer-devel hip-doc rocrand hsa-rocr hipfft hipsparse-devel rocsparse-devel rocrand-devel rocm-opencl hip-devel rocprim-devel hipsolver-devel rocfft-devel hsa-amd-aqlprofile hipify-clang miopen-hip-devel rocm-llvm hip-runtime-amd hip-samples rocalution-devel rccl-devel hipsolver rocprofiler-devel miopen-hip rocm-cmake hipsparse rocblas-devel rocm-opencl-devel + debpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-dev rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-dev hipcub-dev rocsolver rocsparse rocsolver-dev rocminfo hipfft-dev rocm-gdb rocm-dbgapi rocfft hipblas-dev rocthrust-dev openmp-extras-runtime openmp-extras-dev comgr rccl rocblas hipblas roctracer-dev hip-doc rocrand hsa-rocr hipfft hipsparse-dev rocsparse-dev rocrand-dev rocm-opencl hip-dev rocprim-dev hipsolver-dev rocfft-dev hsa-amd-aqlprofile hipify-clang miopen-hip-dev rocm-llvm hip-runtime-amd hip-samples rocalution-dev rccl-dev hipsolver rocprofiler-dev miopen-hip rocm-cmake hipsparse rocblas-dev rocm-opencl-dev + diff --git a/rvs/conf/MI355X/levels/rvs_level_2.conf b/rvs/conf/MI355X/levels/rvs_level_2.conf new file mode 100644 index 000000000..c6bde5336 --- /dev/null +++ b/rvs/conf/MI355X/levels/rvs_level_2.conf @@ -0,0 +1,79 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: pcie_capability + device: all + module: peqt + capability: + link_cap_max_speed: + link_cap_max_width: + link_stat_cur_speed: + link_stat_neg_width: + slot_pwr_limit_value: + slot_physical_num: + deviceid: + vendor_id: + kernel_driver: + +- name: hbm_basic + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 865075200 + test_type: 2 + mibibytes: false + o/p_csv: false + read: true + write: true + subtest: 0 + dwords_per_lane: 4 + chunks_per_block: 1 + tb_size: 512 + +- name: xgmi_interface + device: all + module: pbqt + log_interval: 5000 + duration: 20000 + peers: all + test_bandwidth: true + bidirectional: false + parallel: true + block_size: 67108864 + device_id: all + +- name: pcie_interface + device: all + module: pebb + duration: 20000 + device_to_host: false + host_to_device: true + parallel: true + block_size: 67108864 + link_type: 2 + diff --git a/rvs/conf/MI355X/levels/rvs_level_3.conf b/rvs/conf/MI355X/levels/rvs_level_3.conf new file mode 100644 index 000000000..7ec106c63 --- /dev/null +++ b/rvs/conf/MI355X/levels/rvs_level_3.conf @@ -0,0 +1,248 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: hbm_mid + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 865075200 + test_type: 2 + mibibytes: false + o/p_csv: false + read: true + write: true + copy: true + mul: true + add: true + dwords_per_lane: 4 + chunks_per_block: 1 + tb_size: 512 + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: compute-fp4-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp4_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e3m2_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-bf6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e2m3_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 2061000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 1854000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 946000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 15000 + hot_calls: 10000 + copy_matrix: false + target_stress: 956000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + diff --git a/rvs/conf/MI355X/levels/rvs_level_4.conf b/rvs/conf/MI355X/levels/rvs_level_4.conf new file mode 100644 index 000000000..685d0c18d --- /dev/null +++ b/rvs/conf/MI355X/levels/rvs_level_4.conf @@ -0,0 +1,327 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: hbm_full + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 865075200 + test_type: 2 + mibibytes: false + o/p_csv: false + read: true + write: true + copy: true + mul: true + add: true + triad: true + dot: true + dwords_per_lane: 4 + chunks_per_block: 1 + tb_size: 512 + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: memtest + device: all + module: mem + parallel: true + count: 1 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + +- name: compute-fp4-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp4_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e3m2_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-bf6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e2m3_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 2061000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 1854000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 946000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 956000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-fp32-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: trig + data_type: fp32_r + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-fp64-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp64_r + compute_type: fp64_r + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: power-stress + device: all + module: iet + parallel: true + duration: 60000 + ramp_interval: 1000 + sample_interval: 5000 + log_interval: 5000 + target_power: 1400 + tolerance: 0.01 + bw_workload: true + cp_workload: false + wg_count: 256 + nt_loads: true + diff --git a/rvs/conf/MI355X/levels/rvs_level_5.conf b/rvs/conf/MI355X/levels/rvs_level_5.conf new file mode 100644 index 000000000..eb85e4af5 --- /dev/null +++ b/rvs/conf/MI355X/levels/rvs_level_5.conf @@ -0,0 +1,327 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: hbm_full + device: all + module: babel + parallel: true + count: 1 + num_iter: 10000 + array_size: 865075200 + test_type: 2 + mibibytes: false + o/p_csv: false + read: true + write: true + copy: true + mul: true + add: true + triad: true + dot: true + dwords_per_lane: 4 + chunks_per_block: 1 + tb_size: 512 + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 900000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: memtest + device: all + module: mem + parallel: true + count: 20 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + +- name: compute-fp4-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp4_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e3m2_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-bf6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e2m3_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 10000 + copy_matrix: false + target_stress: 2061000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e4m3_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 10000 + copy_matrix: false + target_stress: 1854000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 10000 + copy_matrix: false + target_stress: 946000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 10000 + copy_matrix: false + target_stress: 956000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-fp32-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: trig + data_type: fp32_r + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-fp64-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp64_r + compute_type: fp64_r + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: power-stress + device: all + module: iet + parallel: true + duration: 3600000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + target_power: 1400 + tolerance: 0.01 + bw_workload: true + cp_workload: false + wg_count: 256 + nt_loads: true + diff --git a/rvs/conf/MI355X/pbqt_single.conf b/rvs/conf/MI355X/pbqt_single.conf new file mode 100644 index 000000000..a63588804 --- /dev/null +++ b/rvs/conf/MI355X/pbqt_single.conf @@ -0,0 +1,108 @@ +# ################################################################################ +# # +# # Copyright (c) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +# Peer-to-Peer/Device-to-Device/GPU-to-GPU unidirectional bandwidth test +# using RVS native method via SDMA +- name: xgmi_d2d_unidir_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: false + parallel: false + block_size: 1073741824 + device_id: all + +# Peer-to-Peer/Device-to-Device/GPU-to-GPU bidirectional bandwidth test +# using RVS native method via SDMA +- name: xgmi_d2d_bidir_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +# Peer-to-Peer/Device-to-Device/GPU-to-GPU unidirectional bandwidth test +# using TransferBench via SDMA +- name: xgmi_d2d_unidir_tb_dma + device: all + module: pbqt + duration: 0 + peers: all + test_bandwidth: true + bidirectional: false + parallel: false + block_size: 1073741824 + device_id: all + transfer_method: transferbench + executor: dma + subexecutor: 1 + warm_calls: 2 + hot_calls: 5 + +# Peer-to-Peer/Device-to-Device/GPU-to-GPU unidirectional bandwidth test +# using TransferBench via GFX +- name: xgmi_d2d_unidir_tb_gfx + device: all + module: pbqt + duration: 0 + peers: all + test_bandwidth: true + bidirectional: false + parallel: false + block_size: 1073741824 + device_id: all + transfer_method: transferbench + executor: gfx + subexecutor: 16 + warm_calls: 2 + hot_calls: 5 + +# All-to-All bandwidth test using TransferBench via GFX +- name: xgmi_d2d_unidir_tb_a2a + device: all + module: pbqt + peers: all + test_bandwidth: true + bidirectional: false + parallel: false + block_size: 268435456 + device_id: all + transfer_method: transferbench + transferbench_test: alltoall + executor: gfx + subexecutor: 8 + warm_calls: 3 + hot_calls: 10 + gfx_unroll: 2 + diff --git a/rvs/conf/MI355X/pebb_single.conf b/rvs/conf/MI355X/pebb_single.conf new file mode 100644 index 000000000..a20fc9789 --- /dev/null +++ b/rvs/conf/MI355X/pebb_single.conf @@ -0,0 +1,120 @@ +# ################################################################################ +# # +# # Copyright (c) 2025-2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +# Host-to-Device/CPU-to-GPU bandwidth test using RVS native method via SDMA +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +# Device-to-Host/GPU-to-CPU bandwidth test using RVS native method via SDMA +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +# Host-to-Device/CPU-to-GPU bandwidth test using TransferBench via SDMA +- name: pcie_h2d_bandwidth_tb_dma + device: all + module: pebb + duration: 0 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + transfer_method: transferbench + executor: dma + subexecutor: 1 + warm_calls: 3 + hot_calls: 10 + source_memory: cpu + destination_memory: gpu + +# Device-to-Host/GPU-to-CPU bandwidth test using TransferBench via SDMA +- name: pcie_d2h_bandwidth_tb_dma + device: all + module: pebb + duration: 0 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + transfer_method: transferbench + executor: dma + subexecutor: 1 + warm_calls: 3 + hot_calls: 10 + source_memory: gpu + destination_memory: cpu + +# Host-to-Device/CPU-to-GPU bandwidth test using TransferBench via GFX +- name: pcie_h2d_bandwidth_tb_gfx + device: all + module: pebb + duration: 0 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + transfer_method: transferbench + executor: gfx + subexecutor: 8 + warm_calls: 3 + hot_calls: 10 + source_memory: cpu + destination_memory: "null" + +# Device-to-Host/GPU-to-CPU bandwidth test using TransferBench via GFX +- name: pcie_d2h_bandwidth_tb_gfx + device: all + module: pebb + duration: 0 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + transfer_method: transferbench + executor: gfx + subexecutor: 8 + warm_calls: 3 + hot_calls: 10 + source_memory: "null" + destination_memory: cpu + diff --git a/rvs/conf/MI450X/babel.conf b/rvs/conf/MI450X/babel.conf new file mode 100644 index 000000000..d79331230 --- /dev/null +++ b/rvs/conf/MI450X/babel.conf @@ -0,0 +1,62 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# BABEL test +# +# Preconditions: +# Set device to all. If you need to run the rvs only on a subset of GPUs, please run rvs with -g +# option, collect the GPUs IDs (e.g.: GPU[ 5 - 50599] -> 50599 is the GPU ID) and then specify +# all the GPUs IDs separated by white space (e.g.: device: 50599 3245) +# Set parallel execution to true +# Set array size to reflect the buffer you want to test +# Set run count to 1 (test will run once) +# + +actions: +- name: babel-float-1950MiB + device: all + module: babel # Name of the module + parallel: true # Parallel true or false + count: 1 # Number of times you want to repeat the test from the begin ( A clean start every time) + num_iter: 20000 # Number of iterations, this many kernels are launched simultaneosuly and stresses the system + array_size: 2044723200 # Array size the test operates, this is 1950 MiB + test_type: 1 # type of test, 1: Float, 2: Double, 3: Triad float, 4: Triad double + mibibytes: false # mibibytes (MiB) or megabytes (MB), true for MiB + o/p_csv: false # o/p as csv file + read: true # read operation + write: false # write operation + copy: false # copy operation + mul: false # mul operation + add: false # add operation + dot: false # dot operation + triad: true # triad operation + dwords_per_lane: 4 # Number of dwords per lane + chunks_per_block: 4 # Number of chunks per block + tb_size: 1024 # Thread block size + data_init: gpu_norm_dist # Data initialization + nontemporal: write # Non-Temporal + sustained: true # sustained bandwidth mode + + diff --git a/rvs/conf/MI300X/gst_ext.conf b/rvs/conf/MI450X/gst_single.conf similarity index 62% rename from rvs/conf/MI300X/gst_ext.conf rename to rvs/conf/MI450X/gst_single.conf index 6be9c7027..bbb4847dc 100644 --- a/rvs/conf/MI300X/gst_ext.conf +++ b/rvs/conf/MI450X/gst_single.conf @@ -1,6 +1,6 @@ # ################################################################################ # # -# # Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. # # # # MIT LICENSE: # # Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -24,71 +24,79 @@ # ############################################################################### actions: -- name: gst-1000Tflops-8KB-fp8_r-false - device: all - module: gst - parallel: false - count: 1 - duration: 30000 - copy_matrix: false - target_stress: 1000000 - matrix_size_a: 8192 - matrix_size_b: 8192 - matrix_size_c: 8192 - data_type: fp8_r - transa: 1 - transb: 0 - alpha: 1 - beta: 0 - -- name: gst-1000Tflops-8KB-fp8_r-true +- name: gst-4K*4K*32K-trig-fp8 device: all module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 parallel: true - count: 1 - duration: 60000 + hot_calls: 60 copy_matrix: false - target_stress: 1000000 - matrix_size_a: 8192 - matrix_size_b: 8192 - matrix_size_c: 8192 - data_type: fp8_r - transa: 1 + target_stress: 0 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 32768 + matrix_init: trig + data_type: fp8_e4m3_r + out_data_type: fp32_r + compute_type: fp32_r + lda: 0 + ldb: 0 + ldc: 0 + ldd: 0 + transa: 0 transb: 0 alpha: 1 beta: 0 + rotating: 512 + blas_source: hipblaslt -- name: gst-500Tflops-4KB-bf16_r-false +- name: gst-4K*4K*16K-trig-bf16 device: all module: gst - parallel: false - count: 1 - duration: 30000 + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + parallel: true + hot_calls: 60 copy_matrix: false - target_stress: 500000 + target_stress: 0 matrix_size_a: 4096 matrix_size_b: 4096 - matrix_size_c: 8192 + matrix_size_c: 16384 + matrix_init: trig data_type: bf16_r - transa: 1 + out_data_type: bf16_r + compute_type: fp32_r + transa: 0 transb: 0 alpha: 1 beta: 0 + rotating: 512 + blas_source: hipblaslt -- name: gst-500Tflops-4KB-bf16_r-true +- name: gst-16*64K*16K-trig-fp16 device: all module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 parallel: true - count: 1 - duration: 60000 + hot_calls: 60 copy_matrix: false - target_stress: 500000 - matrix_size_a: 4096 - matrix_size_b: 4096 - matrix_size_c: 8192 - data_type: bf16_r - transa: 1 + target_stress: 0 + matrix_size_a: 16 + matrix_size_b: 65536 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 0 transb: 0 alpha: 1 beta: 0 + rotating: 512 + blas_source: hipblaslt diff --git a/rvs/conf/MI450X/iet_stress.conf b/rvs/conf/MI450X/iet_stress.conf new file mode 100644 index 000000000..9f5ee1ea5 --- /dev/null +++ b/rvs/conf/MI450X/iet_stress.conf @@ -0,0 +1,64 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# IET stress test +# +# Preconditions: +# Set device to all. If you need to run the rvs only on a subset of GPUs, please run rvs with -g +# option, collect the GPUs IDs (e.g.: GPU[ 5 - 50599] -> 50599 is the GPU ID) and then specify +# all the GPUs IDs separated by comma. +# Set parallel execution to true (gemm workload execution on all GPUs in parallel) +# Set gemm operation type as dgemm. +# Set matrix_size to 28000. +# Test duration set to 10 mins. +# +# Run test with: +# cd bin +# ./rvs -c conf/MI450X/iet_stress.conf +# +# Expected result: +# The test on each GPU passes (TRUE) if the GPU achieves power target. +# + +actions: +- name: iet-stress-dgemm-true + device: all + module: iet + parallel: true + duration: 600000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + target_power: 0 + tolerance: 0 + matrix_size: 28000 + ops_type: dgemm + lda: 28000 + ldb: 28000 + ldc: 28000 + alpha: 1 + beta: 1 + matrix_init: hiprand + diff --git a/rvs/conf/MI450X/levels/rvs_level_1.conf b/rvs/conf/MI450X/levels/rvs_level_1.conf new file mode 100644 index 000000000..501a3770f --- /dev/null +++ b/rvs/conf/MI450X/levels/rvs_level_1.conf @@ -0,0 +1,37 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: + +- name: metapackage-validation + device: all + module: rcqt + package: rocm-hip-sdk rocm-hip-libraries rocm-ml-libraries rocm-ml-sdk amdgpu-dkms rocm-language-runtime rocm-openmp-sdk rocm-utils rocm-opencl-sdk rocm-opencl-runtime rocm-hip-runtime rocm-developer-tools + +- name: packagelist-install-validation + device: all + module: rcqt + rpmpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-devel rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-devel hipcub-devel rocsolver rocsparse rocsolver-devel rocminfo hipfft-devel rocm-gdb rocm-dbgapi rocfft hipblas-devel rocthrust-devel openmp-extras-runtime openmp-extras-devel comgr rccl rocblas hipblas roctracer-devel hip-doc rocrand hsa-rocr hipfft hipsparse-devel rocsparse-devel rocrand-devel rocm-opencl hip-devel rocprim-devel hipsolver-devel rocfft-devel hsa-amd-aqlprofile hipify-clang miopen-hip-devel rocm-llvm hip-runtime-amd hip-samples rocalution-devel rccl-devel hipsolver rocprofiler-devel miopen-hip rocm-cmake hipsparse rocblas-devel rocm-opencl-devel + debpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-dev rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-dev hipcub-dev rocsolver rocsparse rocsolver-dev rocminfo hipfft-dev rocm-gdb rocm-dbgapi rocfft hipblas-dev rocthrust-dev openmp-extras-runtime openmp-extras-dev comgr rccl rocblas hipblas roctracer-dev hip-doc rocrand hsa-rocr hipfft hipsparse-dev rocsparse-dev rocrand-dev rocm-opencl hip-dev rocprim-dev hipsolver-dev rocfft-dev hsa-amd-aqlprofile hipify-clang miopen-hip-dev rocm-llvm hip-runtime-amd hip-samples rocalution-dev rccl-dev hipsolver rocprofiler-dev miopen-hip rocm-cmake hipsparse rocblas-dev rocm-opencl-dev diff --git a/rvs/conf/MI450X/levels/rvs_level_2.conf b/rvs/conf/MI450X/levels/rvs_level_2.conf new file mode 100644 index 000000000..42894d7ae --- /dev/null +++ b/rvs/conf/MI450X/levels/rvs_level_2.conf @@ -0,0 +1,81 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: pcie_capability + device: all + module: peqt + capability: + link_cap_max_speed: + link_cap_max_width: + link_stat_cur_speed: + link_stat_neg_width: + slot_pwr_limit_value: + slot_physical_num: + deviceid: + vendor_id: + kernel_driver: + +- name: hbm_basic + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 2044723200 + test_type: 2 + mibibytes: false + o/p_csv: false + read: true + write: true + subtest: 0 + dwords_per_lane: 4 + chunks_per_block: 4 + tb_size: 1024 + data_init: gpu_norm_dist + nontemporal: write + sustained: true # sustained bandwidth mode + +- name: xgmi_interface + device: all + module: pbqt + log_interval: 5000 + duration: 20000 + peers: all + test_bandwidth: true + bidirectional: false + parallel: true + block_size: 67108864 + device_id: all + +- name: pcie_interface + device: all + module: pebb + duration: 20000 + device_to_host: false + host_to_device: true + parallel: true + block_size: 67108864 + link_type: 0 diff --git a/rvs/conf/MI450X/levels/rvs_level_3.conf b/rvs/conf/MI450X/levels/rvs_level_3.conf new file mode 100644 index 000000000..403b6bf03 --- /dev/null +++ b/rvs/conf/MI450X/levels/rvs_level_3.conf @@ -0,0 +1,254 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: hbm_mid + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 2044723200 + test_type: 2 + mibibytes: false + o/p_csv: false + read: true + write: true + copy: true + mul: true + add: true + dwords_per_lane: 4 + chunks_per_block: 4 + tb_size: 1024 + data_init: gpu_norm_dist + nontemporal: write + sustained: true # sustained bandwidth mode + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 0 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 0 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: compute-fp4-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp4_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e3m2_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-bf6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e2m3_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 60 + copy_matrix: false + target_stress: 0 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 32768 + matrix_init: trig + data_type: fp8_e4m3_r + out_data_type: fp32_r + compute_type: fp32_r + lda: 0 + ldb: 0 + ldc: 0 + ldd: 0 + transa: 0 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 5000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 1854000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 60 + copy_matrix: false + target_stress: 0 + matrix_size_a: 16 + matrix_size_b: 65536 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 0 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 60 + copy_matrix: false + target_stress: 0 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 16384 + matrix_init: trig + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 0 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true diff --git a/rvs/conf/MI450X/levels/rvs_level_4.conf b/rvs/conf/MI450X/levels/rvs_level_4.conf new file mode 100644 index 000000000..0280b20fc --- /dev/null +++ b/rvs/conf/MI450X/levels/rvs_level_4.conf @@ -0,0 +1,337 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: hbm_full + device: all + module: babel + parallel: true + count: 1 + num_iter: 100 + array_size: 2044723200 + test_type: 2 + mibibytes: false + o/p_csv: false + read: true + write: true + copy: true + mul: true + add: true + triad: true + dot: true + dwords_per_lane: 4 + chunks_per_block: 4 + tb_size: 1024 + data_init: gpu_norm_dist + nontemporal: write + sustained: true # sustained bandwidth mode + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 0 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 0 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: memtest + device: all + module: mem + parallel: true + count: 1 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + +- name: compute-fp4-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp4_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e3m2_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-bf6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e2m3_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 60 + copy_matrix: false + target_stress: 0 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 32768 + matrix_init: trig + data_type: fp8_e4m3_r + out_data_type: fp32_r + compute_type: fp32_r + lda: 0 + ldb: 0 + ldc: 0 + ldd: 0 + transa: 0 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 10000 + copy_matrix: false + target_stress: 1854000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 60 + copy_matrix: false + target_stress: 0 + matrix_size_a: 16 + matrix_size_b: 65536 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 0 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 60 + copy_matrix: false + target_stress: 0 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 16384 + matrix_init: trig + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 0 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-fp32-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: trig + data_type: fp32_r + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-fp64-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 10000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp64_r + compute_type: fp64_r + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: power-stress + device: all + module: iet + parallel: true + duration: 60000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + target_power: 0 + tolerance: 0 + matrix_size: 28000 + ops_type: dgemm + lda: 28000 + ldb: 28000 + ldc: 28000 + alpha: 1 + beta: 1 + matrix_init: hiprand diff --git a/rvs/conf/MI450X/levels/rvs_level_5.conf b/rvs/conf/MI450X/levels/rvs_level_5.conf new file mode 100644 index 000000000..d7b861b99 --- /dev/null +++ b/rvs/conf/MI450X/levels/rvs_level_5.conf @@ -0,0 +1,337 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: hbm_full + device: all + module: babel + parallel: true + count: 1 + num_iter: 10000 + array_size: 2044723200 + test_type: 2 + mibibytes: false + o/p_csv: false + read: true + write: true + copy: true + mul: true + add: true + triad: true + dot: true + dwords_per_lane: 4 + chunks_per_block: 4 + tb_size: 1024 + data_init: gpu_norm_dist + nontemporal: write + sustained: true # sustained bandwidth mode + +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 0 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 0 + +- name: xgmi_d2d_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 900000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +- name: memtest + device: all + module: mem + parallel: true + count: 20 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + +- name: compute-fp4-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp4_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e3m2_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-bf6-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + scale_a: block + scale_b: block + matrix_init: trig + data_type: fp6_e2m3_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.5 + beta: 2 + blas_source: hipblaslt + parallel: true + +- name: compute-fp8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 60 + copy_matrix: false + target_stress: 0 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 32768 + matrix_init: trig + data_type: fp8_e4m3_r + out_data_type: fp32_r + compute_type: fp32_r + lda: 0 + ldb: 0 + ldc: 0 + ldd: 0 + transa: 0 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf8-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 10000 + copy_matrix: false + target_stress: 1854000 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp8_e5m2_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 1 + transb: 0 + alpha: 1.000000 + beta: 0.000000 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-fp16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 60 + copy_matrix: false + target_stress: 0 + matrix_size_a: 16 + matrix_size_b: 65536 + matrix_size_c: 16384 + matrix_init: trig + data_type: fp16_r + out_data_type: fp16_r + compute_type: fp32_r + transa: 0 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-bf16-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 60 + copy_matrix: false + target_stress: 0 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 16384 + matrix_init: trig + data_type: bf16_r + out_data_type: bf16_r + compute_type: fp32_r + transa: 0 + transb: 0 + alpha: 1 + beta: 0 + rotating: 512 + blas_source: hipblaslt + parallel: true + +- name: compute-fp32-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 3072 + matrix_size_b: 3072 + matrix_size_c: 3072 + matrix_init: trig + data_type: fp32_r + lda: 3072 + ldb: 3072 + ldc: 3072 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: compute-fp64-trig + device: all + module: gst + log_interval: 3000 + ramp_interval: 2000 + duration: 900000 + hot_calls: 1000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 8192 + matrix_size_b: 8192 + matrix_size_c: 8192 + matrix_init: trig + data_type: fp64_r + compute_type: fp64_r + lda: 8192 + ldb: 8192 + ldc: 8192 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + blas_source: hipblaslt + parallel: true + +- name: power-stress + device: all + module: iet + parallel: true + duration: 3600000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + target_power: 0 + tolerance: 0 + matrix_size: 28000 + ops_type: dgemm + lda: 28000 + ldb: 28000 + ldc: 28000 + alpha: 1 + beta: 1 + matrix_init: hiprand diff --git a/rvs/conf/MI450X/pbqt_single.conf b/rvs/conf/MI450X/pbqt_single.conf new file mode 100644 index 000000000..c903a34a4 --- /dev/null +++ b/rvs/conf/MI450X/pbqt_single.conf @@ -0,0 +1,108 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +# Peer-to-Peer/Device-to-Device/GPU-to-GPU unidirectional bandwidth test +# using RVS native method via SDMA +- name: xgmi_d2d_unidir_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: false + parallel: false + block_size: 1073741824 + device_id: all + +# Peer-to-Peer/Device-to-Device/GPU-to-GPU bidirectional bandwidth test +# using RVS native method via SDMA +- name: xgmi_d2d_bidir_bandwidth + device: all + module: pbqt + log_interval: 5000 + duration: 30000 + peers: all + test_bandwidth: true + bidirectional: true + parallel: false + block_size: 1073741824 + device_id: all + +# Peer-to-Peer/Device-to-Device/GPU-to-GPU unidirectional bandwidth test +# using TransferBench via SDMA +- name: xgmi_d2d_unidir_tb_dma + device: all + module: pbqt + duration: 0 + peers: all + test_bandwidth: true + bidirectional: false + parallel: false + block_size: 1073741824 + device_id: all + transfer_method: transferbench + executor: dma + subexecutor: 1 + warm_calls: 2 + hot_calls: 5 + +# Peer-to-Peer/Device-to-Device/GPU-to-GPU unidirectional bandwidth test +# using TransferBench via GFX +- name: xgmi_d2d_unidir_tb_gfx + device: all + module: pbqt + duration: 0 + peers: all + test_bandwidth: true + bidirectional: false + parallel: false + block_size: 1073741824 + device_id: all + transfer_method: transferbench + executor: gfx + subexecutor: 16 + warm_calls: 2 + hot_calls: 5 + +# All-to-All bandwidth test using TransferBench via GFX +- name: xgmi_d2d_unidir_tb_a2a + device: all + module: pbqt + peers: all + test_bandwidth: true + bidirectional: false + parallel: false + block_size: 268435456 + device_id: all + transfer_method: transferbench + transferbench_test: alltoall + executor: gfx + subexecutor: 8 + warm_calls: 3 + hot_calls: 10 + gfx_unroll: 2 + diff --git a/rvs/conf/MI450X/pebb_single.conf b/rvs/conf/MI450X/pebb_single.conf new file mode 100644 index 000000000..433cec418 --- /dev/null +++ b/rvs/conf/MI450X/pebb_single.conf @@ -0,0 +1,120 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +# Host-to-Device/CPU-to-GPU bandwidth test using RVS native method via SDMA +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 0 + +# Device-to-Host/GPU-to-CPU bandwidth test using RVS native method via SDMA +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 0 + +# Host-to-Device/CPU-to-GPU bandwidth test using TransferBench via SDMA +- name: pcie_h2d_bandwidth_tb_dma + device: all + module: pebb + duration: 0 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 0 + transfer_method: transferbench + executor: dma + subexecutor: 1 + warm_calls: 3 + hot_calls: 10 + source_memory: cpu + destination_memory: gpu + +# Device-to-Host/GPU-to-CPU bandwidth test using TransferBench via SDMA +- name: pcie_d2h_bandwidth_tb_dma + device: all + module: pebb + duration: 0 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 0 + transfer_method: transferbench + executor: dma + subexecutor: 1 + warm_calls: 3 + hot_calls: 10 + source_memory: gpu + destination_memory: cpu + +# Host-to-Device/CPU-to-GPU bandwidth test using TransferBench via GFX +- name: pcie_h2d_bandwidth_tb_gfx + device: all + module: pebb + duration: 0 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 0 + transfer_method: transferbench + executor: gfx + subexecutor: 8 + warm_calls: 3 + hot_calls: 10 + source_memory: cpu + destination_memory: "null" + +# Device-to-Host/GPU-to-CPU bandwidth test using TransferBench via GFX +- name: pcie_d2h_bandwidth_tb_gfx + device: all + module: pebb + duration: 0 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 0 + transfer_method: transferbench + executor: gfx + subexecutor: 8 + warm_calls: 3 + hot_calls: 10 + source_memory: "null" + destination_memory: cpu + diff --git a/rvs/conf/R9600D/gst_single.conf b/rvs/conf/R9600D/gst_single.conf new file mode 100644 index 000000000..fad2a366e --- /dev/null +++ b/rvs/conf/R9600D/gst_single.conf @@ -0,0 +1,129 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + + +# GST test +# +# +# Run test with: +# cd bin +# sudo ./rvs -c conf/R9600D/gst_single.conf -d 3 +# +# Expected result: +# The test on each GPU passes (TRUE) if the GPU achieves target gflops +# and GPU sustains the gflops +# for the rest of the test duration . +# FALSE otherwise + +actions: +- name: gpustress-4K-fp16-false + device: all + module: gst + parallel: false + count: 1 + duration: 15000 + copy_matrix: false + target_stress: 85000 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + hotcalls: 1000 + data_type: fp16_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + transa: 1 + transb: 0 + blas_source: hipblaslt + +- name: gpustress-4K-fp8-true + device: all + module: gst + parallel: true + count: 1 + duration: 15000 + copy_matrix: false + target_stress: 160000 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + hotcalls: 1000 + data_type: fp8_e4m3_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + transa: 1 + transb: 0 + blas_source: hipblaslt + +- name: gpustress-4K-fp8-false + device: all + module: gst + parallel: false + count: 1 + duration: 15000 + copy_matrix: false + target_stress: 160000 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + hotcalls: 1000 + data_type: fp8_e4m3_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + transa: 1 + transb: 0 + blas_source: hipblaslt + +- name: gpustress-4K-i8-false + device: all + module: gst + parallel: false + count: 1 + duration: 15000 + copy_matrix: false + target_stress: 120000 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + hotcalls: 1000 + data_type: i8_r + compute_type: i32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + transa: 1 + transb: 0 + blas_source: hipblaslt + diff --git a/rvs/conf/R9600D/iet_single.conf b/rvs/conf/R9600D/iet_single.conf new file mode 100644 index 000000000..a228380eb --- /dev/null +++ b/rvs/conf/R9600D/iet_single.conf @@ -0,0 +1,108 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: action_1 + device: all + module: iet + parallel: true + count: 1 + wait: 100 + duration: 30000 + ramp_interval: 5000 + sample_interval: 700 + log_interval: 700 + max_violations: 1 + target_power: 150 + tolerance: 0.06 + matrix_size: 10640 + ops_type: sgemm + +- name: action_2 + device: all + module: iet + parallel: true + count: 1 + wait: 100 + duration: 40000 + ramp_interval: 5000 + sample_interval: 1500 + log_interval: 2000 + max_violations: 1 + target_power: 150 + tolerance: 0.2 + blas_source: rocblas + matrix_size: 8640 + ops_type: sgemm + +- name: action_3 + device: all + module: iet + parallel: false + count: 1 + wait: 100 + duration: 30000 + ramp_interval: 5000 + log_interval: 500 + max_violations: 1 + target_power: 150 + data_type: i8_r + compute_type: i32_r + blas_source: hipblaslt + matrix_size_a: 4608 + matrix_size_b: 4608 + matrix_size_c: 4608 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + lda: 4608 + ldb: 4608 + ldc: 4608 + +- name: action_4 + device: all + module: iet + parallel: true + count: 1 + wait: 100 + duration: 25000 + ramp_interval: 5000 + log_interval: 500 + max_violations: 1 + target_power: 150 + data_type: i8_r + compute_type: i32_r + blas_source: hipblaslt + matrix_size_a: 4608 + matrix_size_b: 4608 + matrix_size_c: 4608 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + lda: 4608 + ldb: 4608 + ldc: 4608 diff --git a/rvs/conf/R9600D/levels/rvs_level_1.conf b/rvs/conf/R9600D/levels/rvs_level_1.conf new file mode 100644 index 000000000..da1fe64c4 --- /dev/null +++ b/rvs/conf/R9600D/levels/rvs_level_1.conf @@ -0,0 +1,42 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# R9600D test level 1 - software validation +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from R9600D/gst_single.conf and R9600D/iet_single.conf. +# IET target_power = 150 W. + +actions: + +- name: metapackage-validation + device: all + module: rcqt + package: rocm-hip-sdk rocm-hip-libraries rocm-ml-libraries rocm-ml-sdk amdgpu-dkms rocm-language-runtime rocm-openmp-sdk rocm-utils rocm-opencl-sdk rocm-opencl-runtime rocm-hip-runtime rocm-developer-tools + +- name: packagelist-install-validation + device: all + module: rcqt + rpmpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-devel rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-devel hipcub-devel rocsolver rocsparse rocsolver-devel rocminfo hipfft-devel rocm-gdb rocm-dbgapi rocfft hipblas-devel rocthrust-devel openmp-extras-runtime openmp-extras-devel comgr rccl rocblas hipblas roctracer-devel hip-doc rocrand hsa-rocr hipfft hipsparse-devel rocsparse-devel rocrand-devel rocm-opencl hip-devel rocprim-devel hipsolver-devel rocfft-devel hsa-amd-aqlprofile hipify-clang miopen-hip-devel rocm-llvm hip-runtime-amd hip-samples rocalution-devel rccl-devel hipsolver rocprofiler-devel miopen-hip rocm-cmake hipsparse rocblas-devel rocm-opencl-devel + debpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-dev rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-dev hipcub-dev rocsolver rocsparse rocsolver-dev rocminfo hipfft-dev rocm-gdb rocm-dbgapi rocfft hipblas-dev rocthrust-dev openmp-extras-runtime openmp-extras-dev comgr rccl rocblas hipblas roctracer-dev hip-doc rocrand hsa-rocr hipfft hipsparse-dev rocsparse-dev rocrand-dev rocm-opencl hip-dev rocprim-dev hipsolver-dev rocfft-dev hsa-amd-aqlprofile hipify-clang miopen-hip-dev rocm-llvm hip-runtime-amd hip-samples rocalution-dev rccl-dev hipsolver rocprofiler-dev miopen-hip rocm-cmake hipsparse rocblas-dev rocm-opencl-dev diff --git a/rvs/conf/nv21/pesm_1.conf b/rvs/conf/R9600D/levels/rvs_level_2.conf similarity index 64% rename from rvs/conf/nv21/pesm_1.conf rename to rvs/conf/R9600D/levels/rvs_level_2.conf index ba1631742..ebc1d478b 100644 --- a/rvs/conf/nv21/pesm_1.conf +++ b/rvs/conf/R9600D/levels/rvs_level_2.conf @@ -1,6 +1,6 @@ # ################################################################################ # # -# # Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. # # # # MIT LICENSE: # # Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -23,25 +23,33 @@ # # # ############################################################################### -# PESM test #1 -# -# Preconditions: -# Set device id to an existing AMD deviceid values -# -# Run test with: -# cd bin -# sudo ./rvs -c conf/pesm2.conf -# -# Expected result: -# Test passes without displaying data for any GPUs +# R9600D test level 2 - hardware capability checks +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from R9600D/gst_single.conf and R9600D/iet_single.conf. +# IET target_power = 150 W. + actions: -- name: act1 +- name: pcie_capability device: all - deviceid: 26720 - module: pesm - monitor: true -- name: act2 + module: peqt + capability: + link_cap_max_speed: + link_cap_max_width: + link_stat_cur_speed: + link_stat_neg_width: + slot_pwr_limit_value: + slot_physical_num: + deviceid: + vendor_id: + kernel_driver: + +- name: pcie_interface device: all - debugwait: 3000 - module: pesm - monitor: false + module: pebb + duration: 20000 + device_to_host: false + host_to_device: true + parallel: true + block_size: 67108864 + link_type: 2 + diff --git a/rvs/conf/R9600D/levels/rvs_level_3.conf b/rvs/conf/R9600D/levels/rvs_level_3.conf new file mode 100644 index 000000000..a1f69e611 --- /dev/null +++ b/rvs/conf/R9600D/levels/rvs_level_3.conf @@ -0,0 +1,128 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# R9600D test level 3 - basic performance +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from R9600D/gst_single.conf and R9600D/iet_single.conf. +# IET target_power = 150 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: compute-fp16 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 5000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 85000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp16_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-fp8 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 5000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 160000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp8_e4m3_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-i8 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 5000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 120000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: i8_r + compute_type: i32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + diff --git a/rvs/conf/R9600D/levels/rvs_level_4.conf b/rvs/conf/R9600D/levels/rvs_level_4.conf new file mode 100644 index 000000000..82f2acc74 --- /dev/null +++ b/rvs/conf/R9600D/levels/rvs_level_4.conf @@ -0,0 +1,157 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# R9600D test level 4 - extended performance +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from R9600D/gst_single.conf and R9600D/iet_single.conf. +# IET target_power = 150 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: memtest + device: all + module: mem + parallel: true + count: 1 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 50000 + exclude: 9 10 + +- name: compute-fp16 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 85000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp16_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-fp8 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 160000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp8_e4m3_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-i8 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 120000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: i8_r + compute_type: i32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: power-stress + device: all + module: iet + parallel: true + duration: 60000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + max_violations: 1 + target_power: 150 + tolerance: 0.01 + matrix_size: 10640 + ops_type: sgemm + blas_source: rocblas + diff --git a/rvs/conf/R9600D/levels/rvs_level_5.conf b/rvs/conf/R9600D/levels/rvs_level_5.conf new file mode 100644 index 000000000..baea57331 --- /dev/null +++ b/rvs/conf/R9600D/levels/rvs_level_5.conf @@ -0,0 +1,157 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# R9600D test level 5 - full stress test +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from R9600D/gst_single.conf and R9600D/iet_single.conf. +# IET target_power = 150 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: memtest + device: all + module: mem + parallel: true + count: 20 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + exclude: 9 10 + +- name: compute-fp16 + device: all + module: gst + parallel: true + count: 1 + duration: 900000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 85000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp16_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-fp8 + device: all + module: gst + parallel: true + count: 1 + duration: 900000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 160000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp8_e4m3_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-i8 + device: all + module: gst + parallel: true + count: 1 + duration: 900000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 120000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: i8_r + compute_type: i32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: power-stress + device: all + module: iet + parallel: true + duration: 3600000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + max_violations: 1 + target_power: 150 + tolerance: 0.01 + matrix_size: 10640 + ops_type: sgemm + blas_source: rocblas + diff --git a/rvs/conf/RX9050/gst_single.conf b/rvs/conf/RX9050/gst_single.conf new file mode 100644 index 000000000..c3031ea7f --- /dev/null +++ b/rvs/conf/RX9050/gst_single.conf @@ -0,0 +1,141 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + + +# GST test +# +# Preconditions: +# Set device to all. If you need to run the rvs only on a subset of GPUs, please run rvs with -g +# option, collect the GPUs IDs (e.g.: GPU[ 5 - 50599] -> 50599 is the GPU ID) and then specify +# all the GPUs IDs separated by white space +# Set parallel execution to false +# Set matrix_size to 8640 (for Vega 10 cards). For Vega 20, the recommended matrix_size is 8640 +# Set run count to 2 (each test will run twice) +# Set copy_matrix to false (the matrices will be copied to GPUs only once) +# +# Run test with: +# cd bin +# sudo ./rvs -c conf/gfx1200-xe/gst_single.conf -d 3 +# +# Expected result: +# The test on each GPU passes (TRUE) if the GPU achieves 5000 gflops +# in maximum 7 seconds and then the GPU sustains the gflops +# for the rest of the test duration (total duration is 18 seconds). +# A single Gflops violation (with a 7% tolerance) is allowed. +# FALSE otherwise + +actions: +- name: gpustress-4096-fp32-false + device: all + module: gst + parallel: false + count: 1 + duration: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp32_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: gpustress-4096-fp16-true + device: all + module: gst + parallel: false + count: 2 + duration: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp16_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: gpustress-4096-fp8-false + device: all + module: gst + parallel: false + count: 1 + duration: 10000 + copy_matrix: false + target_stress: 0 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp8_e4m3_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + + +- name: gpustress-4096-i8-false + device: all + module: gst + parallel: false + count: 1 + wait: 100 + duration: 18000 + ramp_interval: 3500 + log_interval: 1000 + max_violations: 1 + copy_matrix: true + target_stress: 0 + data_type: i8_r + compute_type: i32_r + blas_source: hipblaslt + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + diff --git a/rvs/conf/RX9050/iet_single.conf b/rvs/conf/RX9050/iet_single.conf new file mode 100644 index 000000000..d60c77392 --- /dev/null +++ b/rvs/conf/RX9050/iet_single.conf @@ -0,0 +1,119 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: action_1 + device: all + module: iet + parallel: true + count: 1 + wait: 100 + duration: 20000 + ramp_interval: 5000 + sample_interval: 700 + log_interval: 700 + max_violations: 1 + target_power: 90 + tolerance: 0.06 + matrix_size: 8640 + ops_type: sgemm + blas_source: rocblas + +- name: action_2 + device: all + module: iet + parallel: true + count: 1 + wait: 100 + duration: 20000 + ramp_interval: 5000 + sample_interval: 1500 + log_interval: 2000 + max_violations: 1 + target_power: 90 + tolerance: 0.2 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp16_r + compute_type: fp32_r + lda: 4096 + ldb: 4096 + ldc: 4096 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: action_3 + device: all + module: iet + parallel: true + count: 1 + wait: 100 + duration: 20000 + ramp_interval: 5000 + sample_interval: 1500 + log_interval: 2000 + max_violations: 1 + target_power: 90 + tolerance: 0.2 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp8_e4m3_r + compute_type: fp32_r + lda: 4096 + ldb: 4096 + ldc: 4096 + blas_source: hipblaslt + transa: 1 + transb: 0 + + +- name: action_4 + device: all + module: iet + parallel: true + count: 1 + wait: 100 + duration: 20000 + ramp_interval: 5000 + sample_interval: 1500 + log_interval: 2000 + max_violations: 1 + target_power: 90 + tolerance: 0.2 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: i8_r + compute_type: i32_r + lda: 4096 + ldb: 4096 + ldc: 4096 + blas_source: hipblaslt + transa: 1 + transb: 0 + diff --git a/rvs/conf/RX9060/gst_single.conf b/rvs/conf/RX9060/gst_single.conf new file mode 100644 index 000000000..60000e9cc --- /dev/null +++ b/rvs/conf/RX9060/gst_single.conf @@ -0,0 +1,141 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + + +# GST test +# +# Preconditions: +# Set device to all. If you need to run the rvs only on a subset of GPUs, please run rvs with -g +# option, collect the GPUs IDs (e.g.: GPU[ 5 - 50599] -> 50599 is the GPU ID) and then specify +# all the GPUs IDs separated by white space +# Set parallel execution to false +# Set matrix_size to 8640 (for Vega 10 cards). For Vega 20, the recommended matrix_size is 8640 +# Set run count to 2 (each test will run twice) +# Set copy_matrix to false (the matrices will be copied to GPUs only once) +# +# Run test with: +# cd bin +# sudo ./rvs -c conf/gfx1200/gst_single.conf -d 3 +# +# Expected result: +# The test on each GPU passes (TRUE) if the GPU achieves 5000 gflops +# in maximum 7 seconds and then the GPU sustains the gflops +# for the rest of the test duration (total duration is 18 seconds). +# A single Gflops violation (with a 7% tolerance) is allowed. +# FALSE otherwise + +actions: +- name: gpustress-4096-fp32-false + device: all + module: gst + parallel: false + count: 1 + duration: 10000 + copy_matrix: false + target_stress: 6000 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp32_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: gpustress-4096-fp16-true + device: all + module: gst + parallel: false + count: 1 + duration: 10000 + copy_matrix: false + target_stress: 60000 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp16_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: gpustress-4096-fp8-false + device: all + module: gst + parallel: false + count: 1 + duration: 10000 + copy_matrix: false + target_stress: 105000 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp8_e4m3_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + + +- name: gpustress-4096-i8-false + device: all + module: gst + parallel: false + count: 1 + wait: 100 + duration: 18000 + ramp_interval: 7000 + log_interval: 1000 + max_violations: 1 + copy_matrix: true + target_stress: 82000 + data_type: i8_r + compute_type: i32_r + blas_source: hipblaslt + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + transa: 1 + transb: 0 + alpha: 1 + beta: 0 + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + diff --git a/rvs/conf/RX9060/iet_single.conf b/rvs/conf/RX9060/iet_single.conf new file mode 100644 index 000000000..86b7c1375 --- /dev/null +++ b/rvs/conf/RX9060/iet_single.conf @@ -0,0 +1,118 @@ +# ################################################################################ +# # +# # Copyright (c) 2025 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +actions: +- name: action_1 + device: all + module: iet + parallel: true + count: 1 + wait: 100 + duration: 20000 + ramp_interval: 5000 + sample_interval: 700 + log_interval: 700 + max_violations: 1 + target_power: 130 + tolerance: 0.06 + matrix_size: 8640 + ops_type: sgemm + blas_source: rocblas + +- name: action_2 + device: all + module: iet + parallel: true + count: 1 + wait: 100 + duration: 20000 + ramp_interval: 5000 + sample_interval: 1500 + log_interval: 2000 + max_violations: 1 + target_power: 130 + tolerance: 0.2 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + data_type: fp16_r + compute_type: fp32_r + lda: 2080 + ldb: 2080 + ldc: 2080 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: action_3 + device: all + module: iet + parallel: true + count: 1 + wait: 100 + duration: 20000 + ramp_interval: 5000 + sample_interval: 1500 + log_interval: 2000 + max_violations: 1 + target_power: 130 + tolerance: 0.2 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + data_type: fp8_e4m3_r + compute_type: fp32_r + lda: 2080 + ldb: 2080 + ldc: 2080 + blas_source: hipblaslt + transa: 1 + transb: 0 + + +- name: action_4 + device: all + module: iet + parallel: true + count: 1 + wait: 100 + duration: 20000 + ramp_interval: 5000 + sample_interval: 1500 + log_interval: 2000 + max_violations: 1 + target_power: 130 + tolerance: 0.2 + matrix_size_a: 2048 + matrix_size_b: 2048 + matrix_size_c: 2048 + data_type: i8_r + compute_type: i32_r + lda: 2080 + ldb: 2080 + ldc: 2080 + blas_source: hipblaslt + transa: 1 + transb: 0 diff --git a/rvs/conf/RX9060/levels/rvs_level_1.conf b/rvs/conf/RX9060/levels/rvs_level_1.conf new file mode 100644 index 000000000..34e7dd9ec --- /dev/null +++ b/rvs/conf/RX9060/levels/rvs_level_1.conf @@ -0,0 +1,42 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# RX9060 test level 1 - software validation +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from RX9060/gst_single.conf and RX9060/iet_single.conf. +# IET target_power = 130 W. + +actions: + +- name: metapackage-validation + device: all + module: rcqt + package: rocm-hip-sdk rocm-hip-libraries rocm-ml-libraries rocm-ml-sdk amdgpu-dkms rocm-language-runtime rocm-openmp-sdk rocm-utils rocm-opencl-sdk rocm-opencl-runtime rocm-hip-runtime rocm-developer-tools + +- name: packagelist-install-validation + device: all + module: rcqt + rpmpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-devel rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-devel hipcub-devel rocsolver rocsparse rocsolver-devel rocminfo hipfft-devel rocm-gdb rocm-dbgapi rocfft hipblas-devel rocthrust-devel openmp-extras-runtime openmp-extras-devel comgr rccl rocblas hipblas roctracer-devel hip-doc rocrand hsa-rocr hipfft hipsparse-devel rocsparse-devel rocrand-devel rocm-opencl hip-devel rocprim-devel hipsolver-devel rocfft-devel hsa-amd-aqlprofile hipify-clang miopen-hip-devel rocm-llvm hip-runtime-amd hip-samples rocalution-devel rccl-devel hipsolver rocprofiler-devel miopen-hip rocm-cmake hipsparse rocblas-devel rocm-opencl-devel + debpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-dev rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-dev hipcub-dev rocsolver rocsparse rocsolver-dev rocminfo hipfft-dev rocm-gdb rocm-dbgapi rocfft hipblas-dev rocthrust-dev openmp-extras-runtime openmp-extras-dev comgr rccl rocblas hipblas roctracer-dev hip-doc rocrand hsa-rocr hipfft hipsparse-dev rocsparse-dev rocrand-dev rocm-opencl hip-dev rocprim-dev hipsolver-dev rocfft-dev hsa-amd-aqlprofile hipify-clang miopen-hip-dev rocm-llvm hip-runtime-amd hip-samples rocalution-dev rccl-dev hipsolver rocprofiler-dev miopen-hip rocm-cmake hipsparse rocblas-dev rocm-opencl-dev diff --git a/rvs/conf/RX9060/levels/rvs_level_2.conf b/rvs/conf/RX9060/levels/rvs_level_2.conf new file mode 100644 index 000000000..ad2275398 --- /dev/null +++ b/rvs/conf/RX9060/levels/rvs_level_2.conf @@ -0,0 +1,55 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# RX9060 test level 2 - hardware capability checks +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from RX9060/gst_single.conf and RX9060/iet_single.conf. +# IET target_power = 130 W. + +actions: +- name: pcie_capability + device: all + module: peqt + capability: + link_cap_max_speed: + link_cap_max_width: + link_stat_cur_speed: + link_stat_neg_width: + slot_pwr_limit_value: + slot_physical_num: + deviceid: + vendor_id: + kernel_driver: + +- name: pcie_interface + device: all + module: pebb + duration: 20000 + device_to_host: false + host_to_device: true + parallel: true + block_size: 67108864 + link_type: 2 + diff --git a/rvs/conf/RX9060/levels/rvs_level_3.conf b/rvs/conf/RX9060/levels/rvs_level_3.conf new file mode 100644 index 000000000..dc78f0051 --- /dev/null +++ b/rvs/conf/RX9060/levels/rvs_level_3.conf @@ -0,0 +1,153 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# RX9060 test level 3 - basic performance +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from RX9060/gst_single.conf and RX9060/iet_single.conf. +# IET target_power = 130 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: compute-fp32 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 5000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 6000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp32_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-fp16 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 5000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 60000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp16_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-fp8 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 5000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 105000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp8_e4m3_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-i8 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 5000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 82000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: i8_r + compute_type: i32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + diff --git a/rvs/conf/RX9060/levels/rvs_level_4.conf b/rvs/conf/RX9060/levels/rvs_level_4.conf new file mode 100644 index 000000000..ca6ceba67 --- /dev/null +++ b/rvs/conf/RX9060/levels/rvs_level_4.conf @@ -0,0 +1,182 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# RX9060 test level 4 - extended performance +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from RX9060/gst_single.conf and RX9060/iet_single.conf. +# IET target_power = 130 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: memtest + device: all + module: mem + parallel: true + count: 1 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 50000 + exclude: 9 10 + +- name: compute-fp32 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 6000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp32_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-fp16 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 60000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp16_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-fp8 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 105000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp8_e4m3_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-i8 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 82000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: i8_r + compute_type: i32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: power-stress + device: all + module: iet + parallel: true + duration: 60000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + max_violations: 1 + target_power: 130 + tolerance: 0.01 + matrix_size: 8640 + ops_type: sgemm + blas_source: rocblas + diff --git a/rvs/conf/RX9060/levels/rvs_level_5.conf b/rvs/conf/RX9060/levels/rvs_level_5.conf new file mode 100644 index 000000000..1e9d73d32 --- /dev/null +++ b/rvs/conf/RX9060/levels/rvs_level_5.conf @@ -0,0 +1,182 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# RX9060 test level 5 - full stress test +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from RX9060/gst_single.conf and RX9060/iet_single.conf. +# IET target_power = 130 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: memtest + device: all + module: mem + parallel: true + count: 20 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + exclude: 9 10 + +- name: compute-fp32 + device: all + module: gst + parallel: true + count: 1 + duration: 900000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 6000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp32_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-fp16 + device: all + module: gst + parallel: true + count: 1 + duration: 900000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 60000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp16_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-fp8 + device: all + module: gst + parallel: true + count: 1 + duration: 900000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 105000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp8_e4m3_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-i8 + device: all + module: gst + parallel: true + count: 1 + duration: 900000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 82000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: i8_r + compute_type: i32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: power-stress + device: all + module: iet + parallel: true + duration: 3600000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + max_violations: 1 + target_power: 130 + tolerance: 0.01 + matrix_size: 8640 + ops_type: sgemm + blas_source: rocblas + diff --git a/rvs/conf/RX9070/levels/rvs_level_1.conf b/rvs/conf/RX9070/levels/rvs_level_1.conf new file mode 100644 index 000000000..c13c39838 --- /dev/null +++ b/rvs/conf/RX9070/levels/rvs_level_1.conf @@ -0,0 +1,42 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# RX9070 test level 1 - software validation +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from RX9070/gst_single.conf and RX9070/iet_single.conf. +# IET target_power = 220 W. + +actions: + +- name: metapackage-validation + device: all + module: rcqt + package: rocm-hip-sdk rocm-hip-libraries rocm-ml-libraries rocm-ml-sdk amdgpu-dkms rocm-language-runtime rocm-openmp-sdk rocm-utils rocm-opencl-sdk rocm-opencl-runtime rocm-hip-runtime rocm-developer-tools + +- name: packagelist-install-validation + device: all + module: rcqt + rpmpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-devel rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-devel hipcub-devel rocsolver rocsparse rocsolver-devel rocminfo hipfft-devel rocm-gdb rocm-dbgapi rocfft hipblas-devel rocthrust-devel openmp-extras-runtime openmp-extras-devel comgr rccl rocblas hipblas roctracer-devel hip-doc rocrand hsa-rocr hipfft hipsparse-devel rocsparse-devel rocrand-devel rocm-opencl hip-devel rocprim-devel hipsolver-devel rocfft-devel hsa-amd-aqlprofile hipify-clang miopen-hip-devel rocm-llvm hip-runtime-amd hip-samples rocalution-devel rccl-devel hipsolver rocprofiler-devel miopen-hip rocm-cmake hipsparse rocblas-devel rocm-opencl-devel + debpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-dev rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-dev hipcub-dev rocsolver rocsparse rocsolver-dev rocminfo hipfft-dev rocm-gdb rocm-dbgapi rocfft hipblas-dev rocthrust-dev openmp-extras-runtime openmp-extras-dev comgr rccl rocblas hipblas roctracer-dev hip-doc rocrand hsa-rocr hipfft hipsparse-dev rocsparse-dev rocrand-dev rocm-opencl hip-dev rocprim-dev hipsolver-dev rocfft-dev hsa-amd-aqlprofile hipify-clang miopen-hip-dev rocm-llvm hip-runtime-amd hip-samples rocalution-dev rccl-dev hipsolver rocprofiler-dev miopen-hip rocm-cmake hipsparse rocblas-dev rocm-opencl-dev diff --git a/rvs/conf/RX9070/levels/rvs_level_2.conf b/rvs/conf/RX9070/levels/rvs_level_2.conf new file mode 100644 index 000000000..9bc6fe0fd --- /dev/null +++ b/rvs/conf/RX9070/levels/rvs_level_2.conf @@ -0,0 +1,55 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# RX9070 test level 2 - hardware capability checks +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from RX9070/gst_single.conf and RX9070/iet_single.conf. +# IET target_power = 220 W. + +actions: +- name: pcie_capability + device: all + module: peqt + capability: + link_cap_max_speed: + link_cap_max_width: + link_stat_cur_speed: + link_stat_neg_width: + slot_pwr_limit_value: + slot_physical_num: + deviceid: + vendor_id: + kernel_driver: + +- name: pcie_interface + device: all + module: pebb + duration: 20000 + device_to_host: false + host_to_device: true + parallel: true + block_size: 67108864 + link_type: 2 + diff --git a/rvs/conf/RX9070/levels/rvs_level_3.conf b/rvs/conf/RX9070/levels/rvs_level_3.conf new file mode 100644 index 000000000..618710d10 --- /dev/null +++ b/rvs/conf/RX9070/levels/rvs_level_3.conf @@ -0,0 +1,128 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# RX9070 test level 3 - basic performance +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from RX9070/gst_single.conf and RX9070/iet_single.conf. +# IET target_power = 220 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: compute-fp16 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 5000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 120000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp16_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-fp8 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 5000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 220000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp8_e4m3_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-i8 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 5000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 150000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: i8_r + compute_type: i32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + diff --git a/rvs/conf/RX9070/levels/rvs_level_4.conf b/rvs/conf/RX9070/levels/rvs_level_4.conf new file mode 100644 index 000000000..14aab4fa7 --- /dev/null +++ b/rvs/conf/RX9070/levels/rvs_level_4.conf @@ -0,0 +1,157 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# RX9070 test level 4 - extended performance +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from RX9070/gst_single.conf and RX9070/iet_single.conf. +# IET target_power = 220 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: memtest + device: all + module: mem + parallel: true + count: 1 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 50000 + exclude: 9 10 + +- name: compute-fp16 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 120000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp16_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-fp8 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 220000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp8_e4m3_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-i8 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 150000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: i8_r + compute_type: i32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: power-stress + device: all + module: iet + parallel: true + duration: 60000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + max_violations: 1 + target_power: 220 + tolerance: 0.01 + matrix_size: 10640 + ops_type: sgemm + blas_source: rocblas + diff --git a/rvs/conf/RX9070/levels/rvs_level_5.conf b/rvs/conf/RX9070/levels/rvs_level_5.conf new file mode 100644 index 000000000..0ca083287 --- /dev/null +++ b/rvs/conf/RX9070/levels/rvs_level_5.conf @@ -0,0 +1,157 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# RX9070 test level 5 - full stress test +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from RX9070/gst_single.conf and RX9070/iet_single.conf. +# IET target_power = 220 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: memtest + device: all + module: mem + parallel: true + count: 20 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + exclude: 9 10 + +- name: compute-fp16 + device: all + module: gst + parallel: true + count: 1 + duration: 900000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 120000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp16_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-fp8 + device: all + module: gst + parallel: true + count: 1 + duration: 900000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 220000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp8_e4m3_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-i8 + device: all + module: gst + parallel: true + count: 1 + duration: 900000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 150000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: i8_r + compute_type: i32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: power-stress + device: all + module: iet + parallel: true + duration: 3600000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + max_violations: 1 + target_power: 220 + tolerance: 0.01 + matrix_size: 10640 + ops_type: sgemm + blas_source: rocblas + diff --git a/rvs/conf/RX9070GRE/levels/rvs_level_1.conf b/rvs/conf/RX9070GRE/levels/rvs_level_1.conf new file mode 100644 index 000000000..994a55eba --- /dev/null +++ b/rvs/conf/RX9070GRE/levels/rvs_level_1.conf @@ -0,0 +1,42 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# RX9070GRE test level 1 - software validation +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from RX9070GRE/gst_single.conf and RX9070GRE/iet_single.conf. +# IET target_power = 220 W. + +actions: + +- name: metapackage-validation + device: all + module: rcqt + package: rocm-hip-sdk rocm-hip-libraries rocm-ml-libraries rocm-ml-sdk amdgpu-dkms rocm-language-runtime rocm-openmp-sdk rocm-utils rocm-opencl-sdk rocm-opencl-runtime rocm-hip-runtime rocm-developer-tools + +- name: packagelist-install-validation + device: all + module: rcqt + rpmpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-devel rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-devel hipcub-devel rocsolver rocsparse rocsolver-devel rocminfo hipfft-devel rocm-gdb rocm-dbgapi rocfft hipblas-devel rocthrust-devel openmp-extras-runtime openmp-extras-devel comgr rccl rocblas hipblas roctracer-devel hip-doc rocrand hsa-rocr hipfft hipsparse-devel rocsparse-devel rocrand-devel rocm-opencl hip-devel rocprim-devel hipsolver-devel rocfft-devel hsa-amd-aqlprofile hipify-clang miopen-hip-devel rocm-llvm hip-runtime-amd hip-samples rocalution-devel rccl-devel hipsolver rocprofiler-devel miopen-hip rocm-cmake hipsparse rocblas-devel rocm-opencl-devel + debpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-dev rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-dev hipcub-dev rocsolver rocsparse rocsolver-dev rocminfo hipfft-dev rocm-gdb rocm-dbgapi rocfft hipblas-dev rocthrust-dev openmp-extras-runtime openmp-extras-dev comgr rccl rocblas hipblas roctracer-dev hip-doc rocrand hsa-rocr hipfft hipsparse-dev rocsparse-dev rocrand-dev rocm-opencl hip-dev rocprim-dev hipsolver-dev rocfft-dev hsa-amd-aqlprofile hipify-clang miopen-hip-dev rocm-llvm hip-runtime-amd hip-samples rocalution-dev rccl-dev hipsolver rocprofiler-dev miopen-hip rocm-cmake hipsparse rocblas-dev rocm-opencl-dev diff --git a/rvs/conf/RX9070GRE/levels/rvs_level_2.conf b/rvs/conf/RX9070GRE/levels/rvs_level_2.conf new file mode 100644 index 000000000..e9d8f9a1e --- /dev/null +++ b/rvs/conf/RX9070GRE/levels/rvs_level_2.conf @@ -0,0 +1,55 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# RX9070GRE test level 2 - hardware capability checks +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from RX9070GRE/gst_single.conf and RX9070GRE/iet_single.conf. +# IET target_power = 220 W. + +actions: +- name: pcie_capability + device: all + module: peqt + capability: + link_cap_max_speed: + link_cap_max_width: + link_stat_cur_speed: + link_stat_neg_width: + slot_pwr_limit_value: + slot_physical_num: + deviceid: + vendor_id: + kernel_driver: + +- name: pcie_interface + device: all + module: pebb + duration: 20000 + device_to_host: false + host_to_device: true + parallel: true + block_size: 67108864 + link_type: 2 + diff --git a/rvs/conf/RX9070GRE/levels/rvs_level_3.conf b/rvs/conf/RX9070GRE/levels/rvs_level_3.conf new file mode 100644 index 000000000..f962f2df7 --- /dev/null +++ b/rvs/conf/RX9070GRE/levels/rvs_level_3.conf @@ -0,0 +1,128 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# RX9070GRE test level 3 - basic performance +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from RX9070GRE/gst_single.conf and RX9070GRE/iet_single.conf. +# IET target_power = 220 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: compute-fp16 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 5000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 90000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp16_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-fp8 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 5000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 195000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp8_e4m3_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-i8 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 5000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 140000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: i8_r + compute_type: i32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + diff --git a/rvs/conf/RX9070GRE/levels/rvs_level_4.conf b/rvs/conf/RX9070GRE/levels/rvs_level_4.conf new file mode 100644 index 000000000..c9466997a --- /dev/null +++ b/rvs/conf/RX9070GRE/levels/rvs_level_4.conf @@ -0,0 +1,157 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# RX9070GRE test level 4 - extended performance +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from RX9070GRE/gst_single.conf and RX9070GRE/iet_single.conf. +# IET target_power = 220 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: memtest + device: all + module: mem + parallel: true + count: 1 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 50000 + exclude: 9 10 + +- name: compute-fp16 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 90000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp16_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-fp8 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 195000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp8_e4m3_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-i8 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 140000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: i8_r + compute_type: i32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: power-stress + device: all + module: iet + parallel: true + duration: 60000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + max_violations: 1 + target_power: 220 + tolerance: 0.01 + matrix_size: 10640 + ops_type: sgemm + blas_source: rocblas + diff --git a/rvs/conf/RX9070GRE/levels/rvs_level_5.conf b/rvs/conf/RX9070GRE/levels/rvs_level_5.conf new file mode 100644 index 000000000..16d52a2eb --- /dev/null +++ b/rvs/conf/RX9070GRE/levels/rvs_level_5.conf @@ -0,0 +1,157 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# RX9070GRE test level 5 - full stress test +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from RX9070GRE/gst_single.conf and RX9070GRE/iet_single.conf. +# IET target_power = 220 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: memtest + device: all + module: mem + parallel: true + count: 20 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + exclude: 9 10 + +- name: compute-fp16 + device: all + module: gst + parallel: true + count: 1 + duration: 900000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 90000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp16_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-fp8 + device: all + module: gst + parallel: true + count: 1 + duration: 900000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 195000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp8_e4m3_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-i8 + device: all + module: gst + parallel: true + count: 1 + duration: 900000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 140000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: i8_r + compute_type: i32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: power-stress + device: all + module: iet + parallel: true + duration: 3600000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + max_violations: 1 + target_power: 220 + tolerance: 0.01 + matrix_size: 10640 + ops_type: sgemm + blas_source: rocblas + diff --git a/rvs/conf/babel.conf b/rvs/conf/babel.conf index 188135ad1..8d2b368fe 100644 --- a/rvs/conf/babel.conf +++ b/rvs/conf/babel.conf @@ -1,6 +1,6 @@ # ################################################################################ # # -# # Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. +# # Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. # # # # MIT LICENSE: # # Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -40,9 +40,17 @@ actions: module: babel #Name of the module parallel: true # Parallel true or false count: 1 # Number of times you want to repeat the test from the begin ( A clean start every time) - num_iter: 5000 # Number of iterations, this many kernels are launched simultaneosuly and stresses the system + num_iter: 5000 # Number of iterations, this many kernels are launched simultaneously and stresses the system + duration: 0 # Duration in milliseconds. When > 0, runs for this time instead of num_iter. 0 = use num_iter array_size: 33554432 # Buffer size the test operates , this is 32MB test_type: 2 # type of test, 1: Float, 2: Double, 3: Triad float, 4: Triad double mibibytes: false # mibibytes , if you want to specify in bytes (array size in bytes) , make this true o/p_csv: false # o/p as csv file - subtest: 1 # 1: copy 2: copy+mul 3: copy+mul+add 4: copy+mul+add+traid 5: copy+mul+add+traid+dot + copy: true # copy operation + read: true # read operation + write: true # write operation + mul: true # mul operation + add: true # add operation + dot: true # dot operation + triad: true # triad operation + diff --git a/rvs/conf/gfx1200/levels/rvs_level_1.conf b/rvs/conf/gfx1200/levels/rvs_level_1.conf new file mode 100644 index 000000000..a39e1ce39 --- /dev/null +++ b/rvs/conf/gfx1200/levels/rvs_level_1.conf @@ -0,0 +1,42 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# gfx1200 test level 1 - software validation +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from gfx1200/gst_single.conf and gfx1200/iet_single.conf. +# IET target_power = 150 W. + +actions: + +- name: metapackage-validation + device: all + module: rcqt + package: rocm-hip-sdk rocm-hip-libraries rocm-ml-libraries rocm-ml-sdk amdgpu-dkms rocm-language-runtime rocm-openmp-sdk rocm-utils rocm-opencl-sdk rocm-opencl-runtime rocm-hip-runtime rocm-developer-tools + +- name: packagelist-install-validation + device: all + module: rcqt + rpmpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-devel rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-devel hipcub-devel rocsolver rocsparse rocsolver-devel rocminfo hipfft-devel rocm-gdb rocm-dbgapi rocfft hipblas-devel rocthrust-devel openmp-extras-runtime openmp-extras-devel comgr rccl rocblas hipblas roctracer-devel hip-doc rocrand hsa-rocr hipfft hipsparse-devel rocsparse-devel rocrand-devel rocm-opencl hip-devel rocprim-devel hipsolver-devel rocfft-devel hsa-amd-aqlprofile hipify-clang miopen-hip-devel rocm-llvm hip-runtime-amd hip-samples rocalution-devel rccl-devel hipsolver rocprofiler-devel miopen-hip rocm-cmake hipsparse rocblas-devel rocm-opencl-devel + debpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-dev rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-dev hipcub-dev rocsolver rocsparse rocsolver-dev rocminfo hipfft-dev rocm-gdb rocm-dbgapi rocfft hipblas-dev rocthrust-dev openmp-extras-runtime openmp-extras-dev comgr rccl rocblas hipblas roctracer-dev hip-doc rocrand hsa-rocr hipfft hipsparse-dev rocsparse-dev rocrand-dev rocm-opencl hip-dev rocprim-dev hipsolver-dev rocfft-dev hsa-amd-aqlprofile hipify-clang miopen-hip-dev rocm-llvm hip-runtime-amd hip-samples rocalution-dev rccl-dev hipsolver rocprofiler-dev miopen-hip rocm-cmake hipsparse rocblas-dev rocm-opencl-dev diff --git a/rvs/conf/gfx1200/levels/rvs_level_2.conf b/rvs/conf/gfx1200/levels/rvs_level_2.conf new file mode 100644 index 000000000..aec1fa57e --- /dev/null +++ b/rvs/conf/gfx1200/levels/rvs_level_2.conf @@ -0,0 +1,55 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# gfx1200 test level 2 - hardware capability checks +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from gfx1200/gst_single.conf and gfx1200/iet_single.conf. +# IET target_power = 150 W. + +actions: +- name: pcie_capability + device: all + module: peqt + capability: + link_cap_max_speed: + link_cap_max_width: + link_stat_cur_speed: + link_stat_neg_width: + slot_pwr_limit_value: + slot_physical_num: + deviceid: + vendor_id: + kernel_driver: + +- name: pcie_interface + device: all + module: pebb + duration: 20000 + device_to_host: false + host_to_device: true + parallel: true + block_size: 67108864 + link_type: 2 + diff --git a/rvs/conf/gfx1200/levels/rvs_level_3.conf b/rvs/conf/gfx1200/levels/rvs_level_3.conf new file mode 100644 index 000000000..22fff2db7 --- /dev/null +++ b/rvs/conf/gfx1200/levels/rvs_level_3.conf @@ -0,0 +1,153 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# gfx1200 test level 3 - basic performance +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from gfx1200/gst_single.conf and gfx1200/iet_single.conf. +# IET target_power = 150 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: compute-fp32 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 5000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 7000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp32_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-fp16 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 5000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 36000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp16_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-fp8 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 5000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 125000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp8_e4m3_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-i8 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 5000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 95000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: i8_r + compute_type: i32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + diff --git a/rvs/conf/gfx1200/levels/rvs_level_4.conf b/rvs/conf/gfx1200/levels/rvs_level_4.conf new file mode 100644 index 000000000..95f53be7e --- /dev/null +++ b/rvs/conf/gfx1200/levels/rvs_level_4.conf @@ -0,0 +1,182 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# gfx1200 test level 4 - extended performance +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from gfx1200/gst_single.conf and gfx1200/iet_single.conf. +# IET target_power = 150 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: memtest + device: all + module: mem + parallel: true + count: 1 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 50000 + exclude: 9 10 + +- name: compute-fp32 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 7000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp32_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-fp16 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 36000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp16_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-fp8 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 125000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp8_e4m3_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-i8 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 95000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: i8_r + compute_type: i32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: power-stress + device: all + module: iet + parallel: true + duration: 60000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + max_violations: 1 + target_power: 150 + tolerance: 0.01 + matrix_size: 8640 + ops_type: sgemm + blas_source: rocblas + diff --git a/rvs/conf/gfx1200/levels/rvs_level_5.conf b/rvs/conf/gfx1200/levels/rvs_level_5.conf new file mode 100644 index 000000000..adeb7e4e6 --- /dev/null +++ b/rvs/conf/gfx1200/levels/rvs_level_5.conf @@ -0,0 +1,182 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# gfx1200 test level 5 - full stress test +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from gfx1200/gst_single.conf and gfx1200/iet_single.conf. +# IET target_power = 150 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: memtest + device: all + module: mem + parallel: true + count: 20 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + exclude: 9 10 + +- name: compute-fp32 + device: all + module: gst + parallel: true + count: 1 + duration: 900000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 7000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp32_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-fp16 + device: all + module: gst + parallel: true + count: 1 + duration: 900000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 36000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp16_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-fp8 + device: all + module: gst + parallel: true + count: 1 + duration: 900000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 125000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp8_e4m3_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-i8 + device: all + module: gst + parallel: true + count: 1 + duration: 900000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 95000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: i8_r + compute_type: i32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: power-stress + device: all + module: iet + parallel: true + duration: 3600000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + max_violations: 1 + target_power: 150 + tolerance: 0.01 + matrix_size: 8640 + ops_type: sgemm + blas_source: rocblas + diff --git a/rvs/conf/gfx1201/levels/rvs_level_1.conf b/rvs/conf/gfx1201/levels/rvs_level_1.conf new file mode 100644 index 000000000..69b75c072 --- /dev/null +++ b/rvs/conf/gfx1201/levels/rvs_level_1.conf @@ -0,0 +1,42 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# gfx1201 test level 1 - software validation +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from gfx1201/gst_single.conf and gfx1201/iet_single.conf. +# IET target_power = 300 W. + +actions: + +- name: metapackage-validation + device: all + module: rcqt + package: rocm-hip-sdk rocm-hip-libraries rocm-ml-libraries rocm-ml-sdk amdgpu-dkms rocm-language-runtime rocm-openmp-sdk rocm-utils rocm-opencl-sdk rocm-opencl-runtime rocm-hip-runtime rocm-developer-tools + +- name: packagelist-install-validation + device: all + module: rcqt + rpmpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-devel rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-devel hipcub-devel rocsolver rocsparse rocsolver-devel rocminfo hipfft-devel rocm-gdb rocm-dbgapi rocfft hipblas-devel rocthrust-devel openmp-extras-runtime openmp-extras-devel comgr rccl rocblas hipblas roctracer-devel hip-doc rocrand hsa-rocr hipfft hipsparse-devel rocsparse-devel rocrand-devel rocm-opencl hip-devel rocprim-devel hipsolver-devel rocfft-devel hsa-amd-aqlprofile hipify-clang miopen-hip-devel rocm-llvm hip-runtime-amd hip-samples rocalution-devel rccl-devel hipsolver rocprofiler-devel miopen-hip rocm-cmake hipsparse rocblas-devel rocm-opencl-devel + debpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-dev rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-dev hipcub-dev rocsolver rocsparse rocsolver-dev rocminfo hipfft-dev rocm-gdb rocm-dbgapi rocfft hipblas-dev rocthrust-dev openmp-extras-runtime openmp-extras-dev comgr rccl rocblas hipblas roctracer-dev hip-doc rocrand hsa-rocr hipfft hipsparse-dev rocsparse-dev rocrand-dev rocm-opencl hip-dev rocprim-dev hipsolver-dev rocfft-dev hsa-amd-aqlprofile hipify-clang miopen-hip-dev rocm-llvm hip-runtime-amd hip-samples rocalution-dev rccl-dev hipsolver rocprofiler-dev miopen-hip rocm-cmake hipsparse rocblas-dev rocm-opencl-dev diff --git a/rvs/conf/gfx1201/levels/rvs_level_2.conf b/rvs/conf/gfx1201/levels/rvs_level_2.conf new file mode 100644 index 000000000..f15ee308e --- /dev/null +++ b/rvs/conf/gfx1201/levels/rvs_level_2.conf @@ -0,0 +1,55 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# gfx1201 test level 2 - hardware capability checks +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from gfx1201/gst_single.conf and gfx1201/iet_single.conf. +# IET target_power = 300 W. + +actions: +- name: pcie_capability + device: all + module: peqt + capability: + link_cap_max_speed: + link_cap_max_width: + link_stat_cur_speed: + link_stat_neg_width: + slot_pwr_limit_value: + slot_physical_num: + deviceid: + vendor_id: + kernel_driver: + +- name: pcie_interface + device: all + module: pebb + duration: 20000 + device_to_host: false + host_to_device: true + parallel: true + block_size: 67108864 + link_type: 2 + diff --git a/rvs/conf/gfx1201/levels/rvs_level_3.conf b/rvs/conf/gfx1201/levels/rvs_level_3.conf new file mode 100644 index 000000000..14d9e3dae --- /dev/null +++ b/rvs/conf/gfx1201/levels/rvs_level_3.conf @@ -0,0 +1,128 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# gfx1201 test level 3 - basic performance +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from gfx1201/gst_single.conf and gfx1201/iet_single.conf. +# IET target_power = 300 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: compute-fp16 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 5000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 145000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp16_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-fp8 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 5000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 265000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp8_e4m3_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-i8 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 5000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 200000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: i8_r + compute_type: i32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + diff --git a/rvs/conf/gfx1201/levels/rvs_level_4.conf b/rvs/conf/gfx1201/levels/rvs_level_4.conf new file mode 100644 index 000000000..72da42444 --- /dev/null +++ b/rvs/conf/gfx1201/levels/rvs_level_4.conf @@ -0,0 +1,157 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# gfx1201 test level 4 - extended performance +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from gfx1201/gst_single.conf and gfx1201/iet_single.conf. +# IET target_power = 300 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: memtest + device: all + module: mem + parallel: true + count: 1 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 50000 + exclude: 9 10 + +- name: compute-fp16 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 145000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp16_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-fp8 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 265000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp8_e4m3_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-i8 + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 200000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: i8_r + compute_type: i32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: power-stress + device: all + module: iet + parallel: true + duration: 60000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + max_violations: 1 + target_power: 300 + tolerance: 0.01 + matrix_size: 10640 + ops_type: sgemm + blas_source: rocblas + diff --git a/rvs/conf/gfx1201/levels/rvs_level_5.conf b/rvs/conf/gfx1201/levels/rvs_level_5.conf new file mode 100644 index 000000000..bafe71fca --- /dev/null +++ b/rvs/conf/gfx1201/levels/rvs_level_5.conf @@ -0,0 +1,157 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# gfx1201 test level 5 - full stress test +# RDNA4 Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from gfx1201/gst_single.conf and gfx1201/iet_single.conf. +# IET target_power = 300 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: memtest + device: all + module: mem + parallel: true + count: 20 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + exclude: 9 10 + +- name: compute-fp16 + device: all + module: gst + parallel: true + count: 1 + duration: 900000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 145000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp16_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-fp8 + device: all + module: gst + parallel: true + count: 1 + duration: 900000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 265000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: fp8_e4m3_r + compute_type: fp32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: compute-i8 + device: all + module: gst + parallel: true + count: 1 + duration: 900000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 200000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + data_type: i8_r + compute_type: i32_r + lda: 4128 + ldb: 4128 + ldc: 4128 + ldd: 4128 + blas_source: hipblaslt + transa: 1 + transb: 0 + +- name: power-stress + device: all + module: iet + parallel: true + duration: 3600000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + max_violations: 1 + target_power: 300 + tolerance: 0.01 + matrix_size: 10640 + ops_type: sgemm + blas_source: rocblas + diff --git a/rvs/conf/mem.conf b/rvs/conf/mem.conf index 9901c2b63..599c6abf9 100644 --- a/rvs/conf/mem.conf +++ b/rvs/conf/mem.conf @@ -66,3 +66,4 @@ actions: stress: true num_iter: 50000 exclude : 9 10 + diff --git a/rvs/conf/nv21/levels/rvs_level_1.conf b/rvs/conf/nv21/levels/rvs_level_1.conf new file mode 100644 index 000000000..431a80262 --- /dev/null +++ b/rvs/conf/nv21/levels/rvs_level_1.conf @@ -0,0 +1,42 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# nv21 test level 1 - software validation +# Navi Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from nv21/gst_single.conf and nv21/iet_single.conf. +# IET target_power = 127 W. + +actions: + +- name: metapackage-validation + device: all + module: rcqt + package: rocm-hip-sdk rocm-hip-libraries rocm-ml-libraries rocm-ml-sdk amdgpu-dkms rocm-language-runtime rocm-openmp-sdk rocm-utils rocm-opencl-sdk rocm-opencl-runtime rocm-hip-runtime rocm-developer-tools + +- name: packagelist-install-validation + device: all + module: rcqt + rpmpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-devel rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-devel hipcub-devel rocsolver rocsparse rocsolver-devel rocminfo hipfft-devel rocm-gdb rocm-dbgapi rocfft hipblas-devel rocthrust-devel openmp-extras-runtime openmp-extras-devel comgr rccl rocblas hipblas roctracer-devel hip-doc rocrand hsa-rocr hipfft hipsparse-devel rocsparse-devel rocrand-devel rocm-opencl hip-devel rocprim-devel hipsolver-devel rocfft-devel hsa-amd-aqlprofile hipify-clang miopen-hip-devel rocm-llvm hip-runtime-amd hip-samples rocalution-devel rccl-devel hipsolver rocprofiler-devel miopen-hip rocm-cmake hipsparse rocblas-devel rocm-opencl-devel + debpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-dev rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-dev hipcub-dev rocsolver rocsparse rocsolver-dev rocminfo hipfft-dev rocm-gdb rocm-dbgapi rocfft hipblas-dev rocthrust-dev openmp-extras-runtime openmp-extras-dev comgr rccl rocblas hipblas roctracer-dev hip-doc rocrand hsa-rocr hipfft hipsparse-dev rocsparse-dev rocrand-dev rocm-opencl hip-dev rocprim-dev hipsolver-dev rocfft-dev hsa-amd-aqlprofile hipify-clang miopen-hip-dev rocm-llvm hip-runtime-amd hip-samples rocalution-dev rccl-dev hipsolver rocprofiler-dev miopen-hip rocm-cmake hipsparse rocblas-dev rocm-opencl-dev diff --git a/rvs/conf/pesm_1.conf b/rvs/conf/nv21/levels/rvs_level_2.conf similarity index 64% rename from rvs/conf/pesm_1.conf rename to rvs/conf/nv21/levels/rvs_level_2.conf index 9542c578c..5e6e3c1c5 100644 --- a/rvs/conf/pesm_1.conf +++ b/rvs/conf/nv21/levels/rvs_level_2.conf @@ -1,6 +1,6 @@ # ################################################################################ # # -# # Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. # # # # MIT LICENSE: # # Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -23,28 +23,33 @@ # # # ############################################################################### -# PESM test #1 -# -# Preconditions: -# Set device id to an existing AMD deviceid values -# -# Run test with: -# cd bin -# sudo ./rvs -c conf/pesm_1.conf -# -# Expected result: -# Test passes without displaying data for any GPUs -# -# monitor_action - Monitors changes in PCIe link speed and power state changes +# nv21 test level 2 - hardware capability checks +# Navi Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from nv21/gst_single.conf and nv21/iet_single.conf. +# IET target_power = 127 W. + actions: -- name: monitor_action +- name: pcie_capability device: all - module: pesm - monitor: true + module: peqt + capability: + link_cap_max_speed: + link_cap_max_width: + link_stat_cur_speed: + link_stat_neg_width: + slot_pwr_limit_value: + slot_physical_num: + deviceid: + vendor_id: + kernel_driver: -# wait_action - Just debug wait for the duration set and exit -- name: wait_action +- name: pcie_interface device: all - debugwait: 3000 - module: pesm - monitor: false + module: pebb + duration: 20000 + device_to_host: false + host_to_device: true + parallel: true + block_size: 67108864 + link_type: 2 + diff --git a/rvs/conf/nv21/levels/rvs_level_3.conf b/rvs/conf/nv21/levels/rvs_level_3.conf new file mode 100644 index 000000000..de9155127 --- /dev/null +++ b/rvs/conf/nv21/levels/rvs_level_3.conf @@ -0,0 +1,93 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# nv21 test level 3 - basic performance +# Navi Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from nv21/gst_single.conf and nv21/iet_single.conf. +# IET target_power = 127 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: compute-sgemm + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 5000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 6000 + tolerance: 0.07 + matrix_size_a: 8640 + matrix_size_b: 8640 + matrix_size_c: 8640 + ops_type: sgemm + lda: 8640 + ldb: 8640 + ldc: 8640 + +- name: compute-hgemm + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 5000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 35000 + tolerance: 0.07 + matrix_size_a: 6144 + matrix_size_b: 6144 + matrix_size_c: 6144 + ops_type: hgemm + lda: 6144 + ldb: 6144 + ldc: 6144 + diff --git a/rvs/conf/nv21/levels/rvs_level_4.conf b/rvs/conf/nv21/levels/rvs_level_4.conf new file mode 100644 index 000000000..f79e4f895 --- /dev/null +++ b/rvs/conf/nv21/levels/rvs_level_4.conf @@ -0,0 +1,122 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# nv21 test level 4 - extended performance +# Navi Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from nv21/gst_single.conf and nv21/iet_single.conf. +# IET target_power = 127 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: memtest + device: all + module: mem + parallel: true + count: 1 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 50000 + exclude: 9 10 + +- name: compute-sgemm + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 6000 + tolerance: 0.07 + matrix_size_a: 8640 + matrix_size_b: 8640 + matrix_size_c: 8640 + ops_type: sgemm + lda: 8640 + ldb: 8640 + ldc: 8640 + +- name: compute-hgemm + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 35000 + tolerance: 0.07 + matrix_size_a: 6144 + matrix_size_b: 6144 + matrix_size_c: 6144 + ops_type: hgemm + lda: 6144 + ldb: 6144 + ldc: 6144 + +- name: power-stress + device: all + module: iet + parallel: true + duration: 60000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + max_violations: 1 + target_power: 127 + tolerance: 0.01 + matrix_size: 8640 + ops_type: sgemm + blas_source: rocblas + diff --git a/rvs/conf/nv21/levels/rvs_level_5.conf b/rvs/conf/nv21/levels/rvs_level_5.conf new file mode 100644 index 000000000..5cd563e0b --- /dev/null +++ b/rvs/conf/nv21/levels/rvs_level_5.conf @@ -0,0 +1,122 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# nv21 test level 5 - full stress test +# Navi Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from nv21/gst_single.conf and nv21/iet_single.conf. +# IET target_power = 127 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: memtest + device: all + module: mem + parallel: true + count: 20 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + exclude: 9 10 + +- name: compute-sgemm + device: all + module: gst + parallel: true + count: 1 + duration: 900000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 6000 + tolerance: 0.07 + matrix_size_a: 8640 + matrix_size_b: 8640 + matrix_size_c: 8640 + ops_type: sgemm + lda: 8640 + ldb: 8640 + ldc: 8640 + +- name: compute-hgemm + device: all + module: gst + parallel: true + count: 1 + duration: 900000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 35000 + tolerance: 0.07 + matrix_size_a: 6144 + matrix_size_b: 6144 + matrix_size_c: 6144 + ops_type: hgemm + lda: 6144 + ldb: 6144 + ldc: 6144 + +- name: power-stress + device: all + module: iet + parallel: true + duration: 3600000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + max_violations: 1 + target_power: 127 + tolerance: 0.01 + matrix_size: 8640 + ops_type: sgemm + blas_source: rocblas + diff --git a/rvs/conf/nv31/levels/rvs_level_1.conf b/rvs/conf/nv31/levels/rvs_level_1.conf new file mode 100644 index 000000000..cb15998a2 --- /dev/null +++ b/rvs/conf/nv31/levels/rvs_level_1.conf @@ -0,0 +1,42 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# nv31 test level 1 - software validation +# Navi Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from nv31/gst_single.conf and nv31/iet_single.conf. +# IET target_power = 127 W. + +actions: + +- name: metapackage-validation + device: all + module: rcqt + package: rocm-hip-sdk rocm-hip-libraries rocm-ml-libraries rocm-ml-sdk amdgpu-dkms rocm-language-runtime rocm-openmp-sdk rocm-utils rocm-opencl-sdk rocm-opencl-runtime rocm-hip-runtime rocm-developer-tools + +- name: packagelist-install-validation + device: all + module: rcqt + rpmpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-devel rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-devel hipcub-devel rocsolver rocsparse rocsolver-devel rocminfo hipfft-devel rocm-gdb rocm-dbgapi rocfft hipblas-devel rocthrust-devel openmp-extras-runtime openmp-extras-devel comgr rccl rocblas hipblas roctracer-devel hip-doc rocrand hsa-rocr hipfft hipsparse-devel rocsparse-devel rocrand-devel rocm-opencl hip-devel rocprim-devel hipsolver-devel rocfft-devel hsa-amd-aqlprofile hipify-clang miopen-hip-devel rocm-llvm hip-runtime-amd hip-samples rocalution-devel rccl-devel hipsolver rocprofiler-devel miopen-hip rocm-cmake hipsparse rocblas-devel rocm-opencl-devel + debpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-dev rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-dev hipcub-dev rocsolver rocsparse rocsolver-dev rocminfo hipfft-dev rocm-gdb rocm-dbgapi rocfft hipblas-dev rocthrust-dev openmp-extras-runtime openmp-extras-dev comgr rccl rocblas hipblas roctracer-dev hip-doc rocrand hsa-rocr hipfft hipsparse-dev rocsparse-dev rocrand-dev rocm-opencl hip-dev rocprim-dev hipsolver-dev rocfft-dev hsa-amd-aqlprofile hipify-clang miopen-hip-dev rocm-llvm hip-runtime-amd hip-samples rocalution-dev rccl-dev hipsolver rocprofiler-dev miopen-hip rocm-cmake hipsparse rocblas-dev rocm-opencl-dev diff --git a/rvs/conf/nv32/pesm_1.conf b/rvs/conf/nv31/levels/rvs_level_2.conf similarity index 64% rename from rvs/conf/nv32/pesm_1.conf rename to rvs/conf/nv31/levels/rvs_level_2.conf index ba1631742..dc25777a4 100644 --- a/rvs/conf/nv32/pesm_1.conf +++ b/rvs/conf/nv31/levels/rvs_level_2.conf @@ -1,6 +1,6 @@ # ################################################################################ # # -# # Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. # # # # MIT LICENSE: # # Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -23,25 +23,33 @@ # # # ############################################################################### -# PESM test #1 -# -# Preconditions: -# Set device id to an existing AMD deviceid values -# -# Run test with: -# cd bin -# sudo ./rvs -c conf/pesm2.conf -# -# Expected result: -# Test passes without displaying data for any GPUs +# nv31 test level 2 - hardware capability checks +# Navi Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from nv31/gst_single.conf and nv31/iet_single.conf. +# IET target_power = 127 W. + actions: -- name: act1 +- name: pcie_capability device: all - deviceid: 26720 - module: pesm - monitor: true -- name: act2 + module: peqt + capability: + link_cap_max_speed: + link_cap_max_width: + link_stat_cur_speed: + link_stat_neg_width: + slot_pwr_limit_value: + slot_physical_num: + deviceid: + vendor_id: + kernel_driver: + +- name: pcie_interface device: all - debugwait: 3000 - module: pesm - monitor: false + module: pebb + duration: 20000 + device_to_host: false + host_to_device: true + parallel: true + block_size: 67108864 + link_type: 2 + diff --git a/rvs/conf/nv31/levels/rvs_level_3.conf b/rvs/conf/nv31/levels/rvs_level_3.conf new file mode 100644 index 000000000..e8b185292 --- /dev/null +++ b/rvs/conf/nv31/levels/rvs_level_3.conf @@ -0,0 +1,93 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# nv31 test level 3 - basic performance +# Navi Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from nv31/gst_single.conf and nv31/iet_single.conf. +# IET target_power = 127 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: compute-sgemm + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 5000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 10000 + tolerance: 0.07 + matrix_size_a: 8640 + matrix_size_b: 8640 + matrix_size_c: 8640 + ops_type: sgemm + lda: 8640 + ldb: 8640 + ldc: 8640 + +- name: compute-hgemm + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 5000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 41000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + ops_type: hgemm + lda: 4096 + ldb: 4096 + ldc: 4096 + diff --git a/rvs/conf/nv31/levels/rvs_level_4.conf b/rvs/conf/nv31/levels/rvs_level_4.conf new file mode 100644 index 000000000..f3fe38c2d --- /dev/null +++ b/rvs/conf/nv31/levels/rvs_level_4.conf @@ -0,0 +1,122 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# nv31 test level 4 - extended performance +# Navi Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from nv31/gst_single.conf and nv31/iet_single.conf. +# IET target_power = 127 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: memtest + device: all + module: mem + parallel: true + count: 1 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 50000 + exclude: 9 10 + +- name: compute-sgemm + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 10000 + tolerance: 0.07 + matrix_size_a: 8640 + matrix_size_b: 8640 + matrix_size_c: 8640 + ops_type: sgemm + lda: 8640 + ldb: 8640 + ldc: 8640 + +- name: compute-hgemm + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 41000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + ops_type: hgemm + lda: 4096 + ldb: 4096 + ldc: 4096 + +- name: power-stress + device: all + module: iet + parallel: true + duration: 60000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + max_violations: 1 + target_power: 127 + tolerance: 0.01 + matrix_size: 8640 + ops_type: sgemm + blas_source: rocblas + diff --git a/rvs/conf/nv31/levels/rvs_level_5.conf b/rvs/conf/nv31/levels/rvs_level_5.conf new file mode 100644 index 000000000..4a8dc6cc0 --- /dev/null +++ b/rvs/conf/nv31/levels/rvs_level_5.conf @@ -0,0 +1,122 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# nv31 test level 5 - full stress test +# Navi Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from nv31/gst_single.conf and nv31/iet_single.conf. +# IET target_power = 127 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: memtest + device: all + module: mem + parallel: true + count: 20 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + exclude: 9 10 + +- name: compute-sgemm + device: all + module: gst + parallel: true + count: 1 + duration: 900000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 10000 + tolerance: 0.07 + matrix_size_a: 8640 + matrix_size_b: 8640 + matrix_size_c: 8640 + ops_type: sgemm + lda: 8640 + ldb: 8640 + ldc: 8640 + +- name: compute-hgemm + device: all + module: gst + parallel: true + count: 1 + duration: 900000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 41000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + ops_type: hgemm + lda: 4096 + ldb: 4096 + ldc: 4096 + +- name: power-stress + device: all + module: iet + parallel: true + duration: 3600000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + max_violations: 1 + target_power: 127 + tolerance: 0.01 + matrix_size: 8640 + ops_type: sgemm + blas_source: rocblas + diff --git a/rvs/conf/nv32/levels/rvs_level_1.conf b/rvs/conf/nv32/levels/rvs_level_1.conf new file mode 100644 index 000000000..4dd4ccc0d --- /dev/null +++ b/rvs/conf/nv32/levels/rvs_level_1.conf @@ -0,0 +1,42 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + +# nv32 test level 1 - software validation +# Navi Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from nv32/gst_single.conf and nv32/iet_single.conf. +# IET target_power = 127 W. + +actions: + +- name: metapackage-validation + device: all + module: rcqt + package: rocm-hip-sdk rocm-hip-libraries rocm-ml-libraries rocm-ml-sdk amdgpu-dkms rocm-language-runtime rocm-openmp-sdk rocm-utils rocm-opencl-sdk rocm-opencl-runtime rocm-hip-runtime rocm-developer-tools + +- name: packagelist-install-validation + device: all + module: rcqt + rpmpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-devel rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-devel hipcub-devel rocsolver rocsparse rocsolver-devel rocminfo hipfft-devel rocm-gdb rocm-dbgapi rocfft hipblas-devel rocthrust-devel openmp-extras-runtime openmp-extras-devel comgr rccl rocblas hipblas roctracer-devel hip-doc rocrand hsa-rocr hipfft hipsparse-devel rocsparse-devel rocrand-devel rocm-opencl hip-devel rocprim-devel hipsolver-devel rocfft-devel hsa-amd-aqlprofile hipify-clang miopen-hip-devel rocm-llvm hip-runtime-amd hip-samples rocalution-devel rccl-devel hipsolver rocprofiler-devel miopen-hip rocm-cmake hipsparse rocblas-devel rocm-opencl-devel + debpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-dev rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-dev hipcub-dev rocsolver rocsparse rocsolver-dev rocminfo hipfft-dev rocm-gdb rocm-dbgapi rocfft hipblas-dev rocthrust-dev openmp-extras-runtime openmp-extras-dev comgr rccl rocblas hipblas roctracer-dev hip-doc rocrand hsa-rocr hipfft hipsparse-dev rocsparse-dev rocrand-dev rocm-opencl hip-dev rocprim-dev hipsolver-dev rocfft-dev hsa-amd-aqlprofile hipify-clang miopen-hip-dev rocm-llvm hip-runtime-amd hip-samples rocalution-dev rccl-dev hipsolver rocprofiler-dev miopen-hip rocm-cmake hipsparse rocblas-dev rocm-opencl-dev diff --git a/rvs/conf/nv31/pesm_1.conf b/rvs/conf/nv32/levels/rvs_level_2.conf similarity index 64% rename from rvs/conf/nv31/pesm_1.conf rename to rvs/conf/nv32/levels/rvs_level_2.conf index ba1631742..ba1a1d859 100644 --- a/rvs/conf/nv31/pesm_1.conf +++ b/rvs/conf/nv32/levels/rvs_level_2.conf @@ -1,6 +1,6 @@ # ################################################################################ # # -# # Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. # # # # MIT LICENSE: # # Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -23,25 +23,33 @@ # # # ############################################################################### -# PESM test #1 -# -# Preconditions: -# Set device id to an existing AMD deviceid values -# -# Run test with: -# cd bin -# sudo ./rvs -c conf/pesm2.conf -# -# Expected result: -# Test passes without displaying data for any GPUs +# nv32 test level 2 - hardware capability checks +# Navi Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from nv32/gst_single.conf and nv32/iet_single.conf. +# IET target_power = 127 W. + actions: -- name: act1 +- name: pcie_capability device: all - deviceid: 26720 - module: pesm - monitor: true -- name: act2 + module: peqt + capability: + link_cap_max_speed: + link_cap_max_width: + link_stat_cur_speed: + link_stat_neg_width: + slot_pwr_limit_value: + slot_physical_num: + deviceid: + vendor_id: + kernel_driver: + +- name: pcie_interface device: all - debugwait: 3000 - module: pesm - monitor: false + module: pebb + duration: 20000 + device_to_host: false + host_to_device: true + parallel: true + block_size: 67108864 + link_type: 2 + diff --git a/rvs/conf/nv32/levels/rvs_level_3.conf b/rvs/conf/nv32/levels/rvs_level_3.conf new file mode 100644 index 000000000..da59ac6c2 --- /dev/null +++ b/rvs/conf/nv32/levels/rvs_level_3.conf @@ -0,0 +1,93 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# nv32 test level 3 - basic performance +# Navi Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from nv32/gst_single.conf and nv32/iet_single.conf. +# IET target_power = 127 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: compute-sgemm + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 5000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 6000 + tolerance: 0.07 + matrix_size_a: 8640 + matrix_size_b: 8640 + matrix_size_c: 8640 + ops_type: sgemm + lda: 8640 + ldb: 8640 + ldc: 8640 + +- name: compute-hgemm + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 5000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 41000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + ops_type: hgemm + lda: 4096 + ldb: 4096 + ldc: 4096 + diff --git a/rvs/conf/nv32/levels/rvs_level_4.conf b/rvs/conf/nv32/levels/rvs_level_4.conf new file mode 100644 index 000000000..d7f9e2c1d --- /dev/null +++ b/rvs/conf/nv32/levels/rvs_level_4.conf @@ -0,0 +1,122 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# nv32 test level 4 - extended performance +# Navi Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from nv32/gst_single.conf and nv32/iet_single.conf. +# IET target_power = 127 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 30000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: memtest + device: all + module: mem + parallel: true + count: 1 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 50000 + exclude: 9 10 + +- name: compute-sgemm + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 6000 + tolerance: 0.07 + matrix_size_a: 8640 + matrix_size_b: 8640 + matrix_size_c: 8640 + ops_type: sgemm + lda: 8640 + ldb: 8640 + ldc: 8640 + +- name: compute-hgemm + device: all + module: gst + parallel: true + count: 1 + duration: 10000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 41000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + ops_type: hgemm + lda: 4096 + ldb: 4096 + ldc: 4096 + +- name: power-stress + device: all + module: iet + parallel: true + duration: 60000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + max_violations: 1 + target_power: 127 + tolerance: 0.01 + matrix_size: 8640 + ops_type: sgemm + blas_source: rocblas + diff --git a/rvs/conf/nv32/levels/rvs_level_5.conf b/rvs/conf/nv32/levels/rvs_level_5.conf new file mode 100644 index 000000000..593b550b5 --- /dev/null +++ b/rvs/conf/nv32/levels/rvs_level_5.conf @@ -0,0 +1,122 @@ +# ################################################################################ +# # +# # Copyright (c) 2026 Advanced Micro Devices, Inc. All rights reserved. +# # +# # MIT LICENSE: +# # Permission is hereby granted, free of charge, to any person obtaining a copy of +# # this software and associated documentation files (the "Software"), to deal in +# # the Software without restriction, including without limitation the rights to +# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +# # of the Software, and to permit persons to whom the Software is furnished to do +# # so, subject to the following conditions: +# # +# # The above copyright notice and this permission notice shall be included in all +# # copies or substantial portions of the Software. +# # +# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# # SOFTWARE. +# # +# ############################################################################### + + +# nv32 test level 5 - full stress test +# Navi Radeon level suite (no HBM/babel or XGMI/pbqt). +# GST target_stress and IET target_power sourced from nv32/gst_single.conf and nv32/iet_single.conf. +# IET target_power = 127 W. + + +actions: +- name: pcie_d2h_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: true + host_to_device: false + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: pcie_h2d_bandwidth + device: all + module: pebb + duration: 900000 + device_to_host: false + host_to_device: true + parallel: false + block_size: 1073741824 + link_type: 2 + +- name: memtest + device: all + module: mem + parallel: true + count: 20 + wait: 100 + mapped_memory: false + mem_blocks: 128 + num_passes: 500 + thrds_per_blk: 64 + stress: true + num_iter: 200000 + exclude: 9 10 + +- name: compute-sgemm + device: all + module: gst + parallel: true + count: 1 + duration: 900000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 6000 + tolerance: 0.07 + matrix_size_a: 8640 + matrix_size_b: 8640 + matrix_size_c: 8640 + ops_type: sgemm + lda: 8640 + ldb: 8640 + ldc: 8640 + +- name: compute-hgemm + device: all + module: gst + parallel: true + count: 1 + duration: 900000 + ramp_interval: 2000 + log_interval: 3000 + max_violations: 1 + copy_matrix: false + target_stress: 41000 + tolerance: 0.07 + matrix_size_a: 4096 + matrix_size_b: 4096 + matrix_size_c: 4096 + ops_type: hgemm + lda: 4096 + ldb: 4096 + ldc: 4096 + +- name: power-stress + device: all + module: iet + parallel: true + duration: 3600000 + ramp_interval: 10000 + sample_interval: 5000 + log_interval: 5000 + max_violations: 1 + target_power: 127 + tolerance: 0.01 + matrix_size: 8640 + ops_type: sgemm + blas_source: rocblas + diff --git a/rvs/conf/pulse_single.conf b/rvs/conf/pulse_single.conf new file mode 100644 index 000000000..6f59d39b9 --- /dev/null +++ b/rvs/conf/pulse_single.conf @@ -0,0 +1,112 @@ +# ===================================================================== +# PS: BETA VERSION - Not intended for production use. +# Pass/fail criteria are still being tuned and may change between +# releases. Use only for evaluation, bring-up, and lab testing. +# ===================================================================== +# +# GPU Power Pulse Stress Test Configuration +# +# This test creates power fluctuations by alternating between +# high-compute (GEMM) and idle phases at a configurable rate. +# When running in parallel on multiple GPUs, a two-level barrier +# (CPU + GPU fine-grained atomics) synchronizes all GPUs so they +# spike current simultaneously, maximizing stress on the PSU. +# +# TUNING GUIDANCE: +# pulse_rate - Hz. Use 1-5 Hz for visible power deltas. +# Higher rates (>50 Hz) produce phases shorter +# than the GPU power state transition time and +# the SMI power reporting window (~100ms), so +# the high/low readings converge. +# high_phase_ratio - fraction of the period spent under load. +# 0.5 = equal high/low. Lower values give the +# GPU more time to idle and drop power. +# matrix_size - larger matrices draw more power. 8192+ +# recommended for GPUs with >500W TDP. +# +# Failure conditions: +# - Thermal: junction (or edge) temperature exceeds max_temp_c (°C) +# - GEMM compute errors (data corruption) +# - Power supply unable to handle transient spikes +# +# max_temp_c (float, °C): fail the high phase if reported temp exceeds this. +# Default in module is 105 if omitted. Use 0 to disable the thermal check. +# +# verify_mode: GEMM check runs once at end of the action (not every pulse). +# diff/crc/both — for matrix_size > 2048, CPU accuracy is skipped (self-check +# only) because full CPU reference GEMM would stall for a very long time. +# gpu_sync_wait: parsed for compatibility; barrier uses device sync on stream 0. +# + + +actions: +- name: pulse_stress_basic + device: all + module: pulse + parallel: true + duration: 60000 + sample_interval: 100 + log_interval: 5000 + pulse_rate: 2 + high_phase_ratio: 0.5 + ops_type: sgemm + matrix_size: 8192 + matrix_init: trig + alpha: 2.0 + beta: -1.0 + tolerance: 10.0 + workload_iterations: 128 + halt_on_error: false + gpu_sync_wait: 10000 + verify_mode: diff + max_temp_c: 105 + blas_source: hipblaslt + data_type: fp32_r + compute_type: fp32_r + +- name: pulse_stress_aggressive + device: all + module: pulse + parallel: true + duration: 120000 + sample_interval: 50 + log_interval: 5000 + pulse_rate: 1 + high_phase_ratio: 0.6 + ops_type: sgemm + matrix_size: 16384 + alpha: 2.0 + beta: -1.0 + tolerance: 10.0 + workload_iterations: 256 + halt_on_error: true + gpu_sync_wait: 10000 + verify_mode: diff + max_temp_c: 105 + blas_source: hipblaslt + data_type: fp32_r + compute_type: fp32_r + +- name: pulse_stress_fp16 + device: all + module: pulse + parallel: true + duration: 60000 + sample_interval: 100 + log_interval: 5000 + pulse_rate: 2 + high_phase_ratio: 0.5 + ops_type: hgemm + matrix_size: 8192 + matrix_init: trig + alpha: 2.0 + beta: -1.0 + tolerance: 10.0 + workload_iterations: 128 + halt_on_error: false + gpu_sync_wait: 10000 + verify_mode: diff + max_temp_c: 105 + blas_source: hipblaslt + data_type: fp16_r + compute_type: fp32_r diff --git a/rvs/conf/rcqt_single.conf b/rvs/conf/rcqt_single.conf index f87d53362..3661c5a2f 100644 --- a/rvs/conf/rcqt_single.conf +++ b/rvs/conf/rcqt_single.conf @@ -1,6 +1,6 @@ # ################################################################################ # # -# # Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. +# # Copyright (c) 2018-2025 Advanced Micro Devices, Inc. All rights reserved. # # # # MIT LICENSE: # # Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -28,10 +28,10 @@ actions: - name: metapackage-validation device: all module: rcqt - package: rocm-hip-sdk rocm-hip-libraries rocm-ml-libraries rocm-ml-sdk amdgpu-dkms rocm-language-runtime rocm-openmp-sdk rocm-utils rocm-opencl-sdk rocm-opencl-runtime rocm-hip-runtime rocm-developer-tools + package: rocm rocm-developer-tools rocm-openmp rocm-opencl-sdk rocm-hip - name: packagelist-install-validation device: all module: rcqt - rpmpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-devel rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-devel hipcub-devel rocsolver rocsparse rocsolver-devel rocminfo hipfft-devel rocm-gdb rocm-dbgapi rocfft hipblas-devel rocthrust-devel openmp-extras-runtime openmp-extras-devel comgr rccl rocblas hipblas roctracer-devel hip-doc rocrand hsa-rocr hipfft hipsparse-devel rocsparse-devel rocrand-devel rocm-opencl hip-devel rocprim-devel hipsolver-devel rocfft-devel hsa-amd-aqlprofile hipify-clang miopen-hip-devel rocm-llvm hip-runtime-amd hip-samples rocalution-devel rccl-devel hipsolver rocprofiler-devel miopen-hip rocm-cmake hipsparse rocblas-devel rocm-opencl-devel - debpackagelist: rocm-hip-libraries rocm-core rocm-hip-runtime-dev rocm-language-runtime rocm-hip-runtime rocm-hip-sdk rocm-utils rocm-smi-lib rocalution rocm-debug-agent rocm-device-libs hsa-rocr-dev hipcub-dev rocsolver rocsparse rocsolver-dev rocminfo hipfft-dev rocm-gdb rocm-dbgapi rocfft hipblas-dev rocthrust-dev openmp-extras-runtime openmp-extras-dev comgr rccl rocblas hipblas roctracer-dev hip-doc rocrand hsa-rocr hipfft hipsparse-dev rocsparse-dev rocrand-dev rocm-opencl hip-dev rocprim-dev hipsolver-dev rocfft-dev hsa-amd-aqlprofile hipify-clang miopen-hip-dev rocm-llvm hip-runtime-amd hip-samples rocalution-dev rccl-dev hipsolver rocprofiler-dev miopen-hip rocm-cmake hipsparse rocblas-dev rocm-opencl-dev + debpackagelist: amd-smi-lib migraphx migraphx-dev miopen-hip miopen-hip-dev mivisionx mivisionx-dev rocm-cmake rocm-core rocminfo half rocm-llvm rpp rpp-dev roctracer-dev rocprofiler-sdk rocm-dbgapi hsa-amd-aqlprofile rocm-debug-agent rocprofiler-sdk-rocpd hsa-rocr hipblas hipblaslt rccl rocblas rocm-device-libs hipify-clang hipcc composablekernel-dev hiptensor comgr openmp-extras-runtime openmp-extras-dev rocm-language-runtime hip-runtime-amd + rpmpackagelist: amd-smi-lib migraphx migraphx-devel miopen-hip miopen-hip-devel mivisionx mivisionx-devel rocm-cmake rocm-core rocminfo half rocm-llvm rpp rpp-devel roctracer-devel rocprofiler-sdk rocm-dbgapi hsa-amd-aqlprofile rocm-debug-agent rocprofiler-sdk-rocpd hsa-rocr hipblas hipblaslt rccl rocblas rocm-device-libs hipify-clang hipcc composablekernel-devel hiptensor comgr openmp-extras-runtime openmp-extras-devel rocm-language-runtime hip-runtime-amd diff --git a/rvs/conf/rvs.conf b/rvs/conf/rvs.conf index e7bc7657d..3a43009b7 100644 --- a/rvs/conf/rvs.conf +++ b/rvs/conf/rvs.conf @@ -1,6 +1,6 @@ # ################################################################################ # # -# # Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. +# # Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. # # # # MIT LICENSE: # # Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -132,9 +132,3 @@ actions: stress: true num_iter: 50000 exclude : 9 10 - -- name: action_9 - device: all - deviceid: 26720 - module: pesm - monitor: true diff --git a/rvs/include/rvs.h b/rvs/include/rvs.h index 93b7edb4a..5f1574d91 100644 --- a/rvs/include/rvs.h +++ b/rvs/include/rvs.h @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2023 Advanced Micro Devices, Inc. + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -79,14 +79,12 @@ typedef enum { RVS_MODULE_BABEL = 0, /*!< Memory stress Test */ RVS_MODULE_GPUP, /*!< GPU Properties */ RVS_MODULE_GST, /*!< GPU Stress Test */ - RVS_MODULE_IET, /*!< Input EDPp Test */ + RVS_MODULE_IET, /*!< Input EDPp Test */ RVS_MODULE_MEM, /*!< Memory Test */ RVS_MODULE_PEBB, /*!< PCI Express Bandwidth Benchmark */ RVS_MODULE_PEQT, /*!< PCI Express Qualification Tool */ - RVS_MODULE_PESM, /*!< PCI Express State Monitor */ RVS_MODULE_PBQT, /*!< P2P Benchmark and Qualification Tool */ RVS_MODULE_RCQT, /*!< ROCm Configuration Qualification Tool */ - RVS_MODULE_SMQT, /*!< SBIOS Mapping Qualifications Tool */ RVS_MODULE_MAX /*!< No. of RVS modules */ } rvs_module_t; diff --git a/rvs/include/rvsexec.h b/rvs/include/rvsexec.h index 5b7114f6a..a6ea7c835 100644 --- a/rvs/include/rvsexec.h +++ b/rvs/include/rvsexec.h @@ -41,6 +41,14 @@ enum class yaml_data_type_t { YAML_STRING = 1 }; +typedef struct { + std::string name; + std::string module; + std::string category; + bool result; + uint64_t duration; +} exec_action; + /** * @class exec * @ingroup Launcher @@ -87,6 +95,13 @@ class exec { int user_param; /* Number of times to execute the test */ int num_times; + /* Details of all executed actions */ + std::vector action_details; + + bool in_progress; + + void in_progress_thread(exec_action action_info); + }; } // namespace rvs diff --git a/rvs/novernum.config b/rvs/novernum.config index 82c327a3b..9ea911c45 100644 --- a/rvs/novernum.config +++ b/rvs/novernum.config @@ -1,9 +1,6 @@ gpup: libgpup.so peqt: libpeqt.so -pesm: libpesm.so rcqt: librcqt.so -smqt: libsmqt.so -gm: libgm.so gst: libgst.so pbqt: libpbqt.so pebb: libpebb.so diff --git a/rvs/src/rvs_interface.cpp b/rvs/src/rvs_interface.cpp index be112ec8b..c174e2251 100644 --- a/rvs/src/rvs_interface.cpp +++ b/rvs/src/rvs_interface.cpp @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2023 Advanced Micro Devices, Inc. + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -195,10 +195,8 @@ rvs_status_t rvs_session_execute(rvs_session_id_t session_id) { "mem", "pebb", "peqt", - "pesm", "pbqt", - "rcqt", - "smqt"}; + "rcqt"}; opt.insert({"module", module[rvs_session[session_idx].property.default_conf.module]}); diff --git a/rvs/src/rvscli.cpp b/rvs/src/rvscli.cpp index 2ff9da5a7..f07ce69a0 100644 --- a/rvs/src/rvscli.cpp +++ b/rvs/src/rvscli.cpp @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -151,6 +151,10 @@ void rvs::cli::init_grammar() { grammar.insert(gpair("-i", sp)); grammar.insert(gpair("--indexes", sp)); + sp = std::make_shared("-s", command, value); + grammar.insert(gpair("-s", sp)); + grammar.insert(gpair("--selectActions", sp)); + sp = std::make_shared("-j", command, optionalvalue); grammar.insert(gpair("-j", sp)); grammar.insert(gpair("--json", sp)); @@ -171,8 +175,19 @@ void rvs::cli::init_grammar() { grammar.insert(gpair("-p", sp)); grammar.insert(gpair("--parallel", sp)); - sp = std::make_shared("-t", command); + sp = std::make_shared("-m", command, value); + grammar.insert(gpair("-m", sp)); + grammar.insert(gpair("--module", sp)); + + sp = std::make_shared("-r", command, value); + grammar.insert(gpair("-r", sp)); + grammar.insert(gpair("--run", sp)); + + sp = std::make_shared("-t", command, value); grammar.insert(gpair("-t", sp)); + grammar.insert(gpair("--duration", sp)); + + sp = std::make_shared("--listTests", command); grammar.insert(gpair("--listTests", sp)); sp = std::make_shared("-v", command); diff --git a/rvs/src/rvsexec.cpp b/rvs/src/rvsexec.cpp index 748de2b8e..1f3bf8d05 100644 --- a/rvs/src/rvsexec.cpp +++ b/rvs/src/rvsexec.cpp @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -38,9 +38,7 @@ #include "include/rvsliblogger.h" #include "include/rvsoptions.h" #include "include/rvstrace.h" -#ifdef FETCH_ROCMPATH_FROM_ROCMCORE -#include "rocm-core/rocm_getpath.h" -#endif +#include "include/rvs_util.h" #define MODULE_NAME_CAPS "CLI" @@ -127,34 +125,155 @@ int rvs::exec::run() { if (rvs::options::has_option("-j", &s_json_log_file)) { logger::to_json(true); logger::set_json_log_file(s_json_log_file); - } - // check -c option string config_file; - if (rvs::options::has_option("-c", &val)) { - config_file = val; - } else { - config_file = "../share/rocm-validation-suite/conf/rvs.conf"; + // check -r option + if (rvs::options::has_option("-r", &val)) { + + std::string plaftorm_name = get_gpu_name(); + + if(val != "1" && val != "2" && val != "3" && val != "4" && val != "5") { + rvs::logger::Err("Test level must be between 1 and 5 !", MODULE_NAME_CAPS); + return -1; + } + + if(plaftorm_name.empty()) { + rvs::logger::Err("No test levels support for the detected GPU or no GPU detected !", MODULE_NAME_CAPS); + return -1; + } + + config_file = "../share/rocm-validation-suite/conf/" + plaftorm_name + "/levels/rvs_level_" + val + ".conf"; + // Check if pConfig file exist if not use old path for backward compatibility std::ifstream file(path + config_file); if (!file.good()) { - config_file = "conf/rvs.conf"; + config_file = "conf/" + plaftorm_name + "/levels/rvs_level_" + val + ".conf"; } file.close(); config_file = path + config_file; + + // Check if pConfig file exists + file.open(config_file); + if (!file.good()) { + char buff[1024]; + snprintf(buff, sizeof(buff), + "%s file is missing.", config_file.c_str()); + rvs::logger::Err(buff, MODULE_NAME_CAPS); + return -1; + } else { + file.close(); + } } - // Check if pConfig file exists - std::ifstream file(config_file); - if (!file.good()) { - char buff[1024]; - snprintf(buff, sizeof(buff), - "%s file is missing.", config_file.c_str()); - rvs::logger::Err(buff, MODULE_NAME_CAPS); - return -1; - } else { - file.close(); + // check -m option + string module; + if (rvs::options::has_option("-m", &module)) { + + std::map module_config_map = { + {"babel", "babel.conf"}, + {"gpup", "gpup_single.conf"}, + {"gst", "gst_single.conf"}, + {"iet", "iet_stress.conf"}, + {"mem", "mem.conf"}, + {"pebb", "pebb_single.conf"}, + {"peqt", "peqt_single.conf"}, + {"pbqt", "pbqt_single.conf"}, + {"rcqt", "rcqt_single.conf"}, + {"tst", "tst_single.conf"} + }; + + auto itr = module_config_map.find(module); + if (itr == module_config_map.end()) { + char buff[1024]; + snprintf(buff, sizeof(buff), + "Invalid module name: %s", module.c_str()); + rvs::logger::Err(buff, MODULE_NAME_CAPS); + return -1; + } + + std::string platform_name = get_gpu_name(); + if (platform_name.empty()) { + rvs::logger::Err("No platform support for the detected GPU or no GPU detected !", MODULE_NAME_CAPS); + return -1; + } + + std::vector confs = {itr->second}; + if (module == "iet") { + confs.push_back("iet_single.conf"); + } + + bool found = false; + for (const auto& conf : confs) { + + /* Plaform specific module conf. from RVS installation path */ + config_file = "../share/rocm-validation-suite/conf/" + platform_name + "/" + conf; + + std::ifstream file(path + config_file); + if (!file.good()) { + /* Plaform specific module conf. from RVS local build path */ + config_file = "conf/" + platform_name + "/" + conf; + } + file.close(); + config_file = path + config_file; + + file.open(config_file); + if (file.good()) { + file.close(); + found = true; + break; + } + else { + /* Try default module conf. */ + config_file = path + "conf/" + conf; + file.open(config_file); + if (file.good()) { + file.close(); + found = true; + break; + } + } + file.close(); + } + + if (!found) { + char buff[1024]; + snprintf(buff, sizeof(buff), + "%s file is missing.", config_file.c_str()); + rvs::logger::Err(buff, MODULE_NAME_CAPS); + return -1; + } + } + + if (!rvs::options::has_option("-r", &val) && !rvs::options::has_option("-m")) { + + // check -c option + if (rvs::options::has_option("-c", &val)) { + + config_file = val; + + } else { + config_file = "../share/rocm-validation-suite/conf/rvs.conf"; + // Check if pConfig file exist if not use old path for backward compatibility + std::ifstream file(path + config_file); + if (!file.good()) { + config_file = "conf/rvs.conf"; + } + file.close(); + config_file = path + config_file; + } + + // Check if pConfig file exists + std::ifstream file(config_file); + if (!file.good()) { + char buff[1024]; + snprintf(buff, sizeof(buff), + "%s file is missing.", config_file.c_str()); + rvs::logger::Err(buff, MODULE_NAME_CAPS); + return -1; + } else { + file.close(); + } } // check -n options @@ -171,6 +290,24 @@ int rvs::exec::run() { } } + // check -t option + if (rvs::options::has_option("-t", &val)) { + try { + int64_t dur = std::stoll(val); + if (dur <= 0) { + rvs::logger::Err("duration must be greater than 0", MODULE_NAME_CAPS); + return -1; + } + } + catch(...) { + char buff[1024]; + snprintf(buff, sizeof(buff), + "duration value not a valid integer: %s", val.c_str()); + rvs::logger::Err(buff, MODULE_NAME_CAPS); + return -1; + } + } + // construct modules configuration file relative path val = path + "../share/rocm-validation-suite/conf/.rvsmodules.config"; // Check if config file exists if not check the old file location for backward compatibility @@ -184,7 +321,7 @@ int rvs::exec::run() { return 1; } - if (rvs::options::has_option("-t", &val)) { + if (rvs::options::has_option("--listTests", &val)) { cout << endl << "ROCm Validation Suite (version " << RVS_VERSION_STRING << ")" << endl << endl; cout << "Modules available:" << endl; @@ -276,24 +413,6 @@ int rvs::exec::run(std::map& opt) { string module; string config; yaml_data_type_t data_type; -#ifdef FETCH_ROCMPATH_FROM_ROCMCORE - char *installPath = nullptr; - unsigned int installPathLen = 0; - string rocmPath; - PathErrors_t retVal = PathSuccess; - // Get ROCM install path - retVal = getROCmInstallPath( &installPath, &installPathLen ); - if(retVal == PathSuccess){ - rocmPath = installPath; - } - else { - std::cout << "Failed to get ROCm Install Path: " << retVal <<"\nSet ROCM_PATH in env" << std::endl; - } - // free allocated memory - if(installPath != nullptr) { - free(installPath); - } -#endif options::has_option("pwd", &path); logger::log_level(rvs::logerror); @@ -302,7 +421,7 @@ int rvs::exec::run(std::map& opt) { } else if (rvs::options::has_option(opt, "module", &module)) { -#define RVS_MODULE_MAX 11 +#define RVS_SESSION_MODULE_MAX 9 std::map module_map = { {"babel", 0}, {"gpup", 1}, @@ -311,12 +430,10 @@ int rvs::exec::run(std::map& opt) { {"mem", 4}, {"pebb", 5}, {"peqt", 6}, - {"pesm", 7}, - {"pbqt", 8}, - {"rcqt", 9}, - {"smqt", 10}}; + {"pbqt", 7}, + {"rcqt", 8}}; - string module_config_file[RVS_MODULE_MAX] = + string module_config_file[RVS_SESSION_MODULE_MAX] = { "babel.conf", "gpup_single.conf", @@ -325,16 +442,17 @@ int rvs::exec::run(std::map& opt) { "mem.conf", "pebb_single.conf", "peqt_single.conf", - "pesm_1.conf", "pbqt_single.conf", - "rcqt_single.conf", - "smqt_single.conf" + "rcqt_single.conf" }; auto itr = module_map.find(module); + if (itr == module_map.end()) { + return -1; + } int module_index = itr->second; - if(RVS_MODULE_MAX <= module_index) { + if (RVS_SESSION_MODULE_MAX <= module_index) { return -1; } @@ -345,13 +463,9 @@ int rvs::exec::run(std::map& opt) { config = "conf/" + module_config_file[module_index]; std::ifstream file(path + config); if (!file.good()) { - // configuration file exist in ROCM path ? -#ifdef FETCH_ROCMPATH_FROM_ROCMCORE - path = rocmPath; -#else - path = ROCM_PATH; -#endif - config = "/share/rocm-validation-suite/conf/" + module_config_file[module_index]; + // RVS data dir: from rvs binary path or build-time RVS_DATA_ROOT + path = rvs_get_rvs_data_root_string(); + config = "/conf/" + module_config_file[module_index]; } file.close(); } @@ -387,13 +501,8 @@ int rvs::exec::run(std::map& opt) { val = path + ".rvsmodules.config"; std::ifstream conf_file(val); if (!conf_file.good()) { - // Modules config. file exist in ROCM path ? -#ifdef FETCH_ROCMPATH_FROM_ROCMCORE - path = rocmPath; -#else - path = ROCM_PATH; -#endif - val = path + "/share/rocm-validation-suite/conf/.rvsmodules.config"; + path = rvs_get_rvs_data_root_string(); + val = path + "/conf/.rvsmodules.config"; } } conf_file.close(); @@ -451,6 +560,9 @@ void rvs::exec::do_help() { cout << "-c --config Specify the test configuration file to use.\n\n"; + cout << "-r --run Specify the test level to run. Valid range is 1 to 5,\n"; + cout << " with 5 indicating the highest stress test level.\n\n"; + cout << "-d --debugLevel Specify the debug level for the output log. The range is 0-5 with\n"; cout << " 5 being the highest verbose level.\n\n"; @@ -465,10 +577,20 @@ void rvs::exec::do_help() { cout << " if a path follows this argument, that will be used as json log file\n"; cout << " else a file created in /var/tmp/ with timestamp in name.\n\n"; + cout << "-s --selectActions Comma separated list of action names or 0-based action index numbers\n"; + cout << " to run from the configuration file. Only the matching actions will\n"; + cout << " be executed. All other actions are skipped.\n\n"; + cout << "-l --debugLogFile Generate log file with output and debug information.\n\n"; - - cout << "-t --listTests List the test modules present in RVS.\n\n"; + cout << "-m --module Specify a module name to run the corresponding platform-specific\n"; + cout << " (MI-series GPUs) module configuration file. Valid modules: babel, gpup, gst, iet,\n"; + cout << " mem, pebb, peqt, pbqt, rcqt.\n\n"; + + cout << "-t --duration Specify the test duration (in seconds) for each action.\n"; + cout << " Overrides the duration value in all actions of the configuration file.\n\n"; + + cout << " --listTests List the test modules present in RVS.\n\n"; cout << "-v --verbose Enable detailed logging. Equivalent to specifying -d 5 option.\n\n"; @@ -493,41 +615,8 @@ void rvs::exec::do_help() { int rvs::exec::do_gpu_list() { cout << "\nROCm Validation Suite (version " << RVS_VERSION_STRING << ")\n\n"; - // create action excutor in .so - rvs::action* pa = module::action_create("pesm"); - if (!pa) { - rvs::logger::Err("could not list GPUs.", MODULE_NAME_CAPS); - return 1; - } - - // obtain interface to set parameters and execute action - if1* pif1 = static_cast(pa->get_interface(1)); - if (!pif1) { - rvs::logger::Err("could not obtain interface if1.", MODULE_NAME_CAPS); - module::action_destroy(pa); - return 1; - } - - pif1->property_set("name", "(launcher)"); - - // specify "list GPUs" action - pif1->property_set("do_gpu_list", ""); - - // set command line options: - for (auto clit = rvs::options::get().begin(); - clit != rvs::options::get().end(); ++clit) { - string p(clit->first); - p = "cli." + p; - pif1->property_set(p, clit->second); - } - - // execute action - int sts = pif1->run(); - - // procssing finished, release action object - module::action_destroy(pa); - - return sts; + rvs::gpulist::Initialize(); + return display_gpu_info(get_gpu_info()); } void rvs::exec::action_callback(const action_result_t * result, void * user_param) { diff --git a/rvs/src/rvsexec_do_yaml.cpp b/rvs/src/rvsexec_do_yaml.cpp index 94446d473..107900a09 100644 --- a/rvs/src/rvsexec_do_yaml.cpp +++ b/rvs/src/rvsexec_do_yaml.cpp @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2025 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -26,6 +26,12 @@ #include #include #include +#include +#include +#include +#include +#include +#include #include "include/rvsexec.h" #include "yaml-cpp/yaml.h" @@ -37,9 +43,31 @@ #include "include/rvsliblogger.h" #include "include/rvsoptions.h" #include "include/rvs_util.h" +#include "include/gpu_util.h" + +#ifdef FETCH_ROCMPATH_FROM_ROCMCORE +#include "rocm-core/rocm_version.h" +#endif #define MODULE_NAME_CAPS "CLI" +#ifdef FETCH_ROCMPATH_FROM_ROCMCORE +/** + * C API getROCmVersion (rocm_version.h) must be called from a function declared + * before the C++ getROCmVersion in this file so the name does not self-resolve. + */ +static bool rvsGetROCmVersionStringFromRcmcore(std::string* out) { + unsigned int mj = 0, mn = 0, p = 0; + if (getROCmVersion(&mj, &mn, &p) != VerSuccess) { + return false; + } + std::ostringstream oss; + oss << mj << "." << mn << "." << p; + *out = oss.str(); + return true; +} +#endif + /*** Example rvs.conf file structure actions: @@ -77,13 +105,53 @@ int rvs::exec::do_yaml(const std::string& config_file) { return -1; } + std::set selected_action_names; + std::set selected_action_indices; + std::string select_val; + if (rvs::options::has_option("-s", &select_val) && !select_val.empty()) { + std::replace(select_val.begin(), select_val.end(), ',', ' '); + std::istringstream iss(select_val); + std::vector tokens; + std::string token; + bool all_numeric = true; + while (iss >> token) { + tokens.push_back(token); + try { + size_t pos = 0; + int idx = std::stoi(token, &pos); + if (pos != token.size() || idx < 0) + all_numeric = false; + } catch (...) { + all_numeric = false; + } + } + for (const auto& t : tokens) { + if (all_numeric) + selected_action_indices.insert(std::stoi(t)); + else + selected_action_names.insert(t); + } + } + bool has_selection = !selected_action_names.empty() || + !selected_action_indices.empty(); + /* Number of times to repeat the test */ for (int i = 0; i < num_times; i++) { + int action_idx = 0; // for all actions... - for (YAML::const_iterator it = actions.begin(); it != actions.end(); ++it) { + for (YAML::const_iterator it = actions.begin(); it != actions.end(); + ++it, ++action_idx) { const YAML::Node& action = *it; + if (has_selection) { + std::string action_name = action["name"].as(); + if (selected_action_names.find(action_name) == selected_action_names.end() && + selected_action_indices.find(action_idx) == selected_action_indices.end()) { + continue; + } + } + sts = 0; rvs::logger::log("Action name :" + action["name"].as(), rvs::logresults); @@ -169,6 +237,317 @@ int rvs::exec::do_yaml(const std::string& config_file) { return 0; } +/** + * @brief Action test in progress spinner thread + */ +void rvs::exec::in_progress_thread(exec_action action_info) { + + const char boundary = '|'; + const int columnWidth = 14; + const int actionColumnWidth = 32; + + const string spinner[] = {"||||", "////", "----", "\\\\\\\\"}; + int spinnerIndex = 0; + + while (in_progress) { + + std::cout << "\r" << boundary << " " + << std::setw(actionColumnWidth) << std::left << action_info.name + << " | " << std::setw(columnWidth) << std::left << action_info.module + << " | " << std::setw(columnWidth) << std::left << spinner[spinnerIndex++] + << " " << boundary << std::flush; + + spinnerIndex %= 4; + + std::this_thread::sleep_for(std::chrono::milliseconds(200)); + } +} + +/** + * @brief Get installed ROCm version + */ +std::string getROCmVersion(void) { + + std::string line = "N/A"; +#ifdef FETCH_ROCMPATH_FROM_ROCMCORE + if (rvsGetROCmVersionStringFromRcmcore(&line)) { + return line; + } + return "N/A"; +#else + const std::string vfile = rvs_get_rocm_install_path_string() + "/.info/version-rocm"; + std::ifstream inputFile(vfile); + if (!inputFile) { + return line; + } + std::getline(inputFile, line); + inputFile.close(); + if (line.empty()) { + line = "N/A"; + } + return line; +#endif +} + +/** + * @brief Get operating system name & version. + */ +std::string getOSNameVersion(void) { + + std::ifstream file("/etc/os-release"); + std::string line; + std::string osName; + + if (!file.is_open()) { + return "N/A"; + } + + while (std::getline(file, line)) { + + if (line.find("PRETTY_NAME=") == 0) { + /* Remove the key and quotes */ + + /* Skip "PRETTY_NAME=" */ + osName = line.substr(12); + if (!osName.empty() && osName.front() == '"' && osName.back() == '"') { + /* Remove surrounding quotes */ + osName = osName.substr(1, osName.size() - 2); + + /* OS name till braces */ + size_t pos = osName.find('('); + if (pos != std::string::npos) { + osName = osName.substr(0, pos); + } + } + break; + } + } + + file.close(); + return osName.empty() ? "N/A" : osName; +} + + +/** + * @brief Get amdgpu driver version. + */ +std::string getAmdGpuDriverVersion() { + + std::array buffer; + std::string result; + + // Run the command + std::unique_ptr pipe(popen("dkms status 2>/dev/null", "r"), pclose); + if (!pipe) { + return "N/A"; + } + + // Read the output + while (fgets(buffer.data(), buffer.size(), pipe.get()) != nullptr) { + result += buffer.data(); + } + + // Example output: "amdgpu/6.14.14-2204008.22.04, 6.8.0-59-generic, x86_64: installed " + // Extract version (e.g., 6.14.14) + size_t slashPos = result.find('/'); + size_t commaPos = result.find('-', slashPos); + if (slashPos != std::string::npos && commaPos != std::string::npos) { + return result.substr(slashPos + 1, commaPos - slashPos - 1); + } + + return "N/A"; +} + +/** + * @brief Print system overview details + */ +void systemOverview() { + + // Header column + std::string header = "System Overview"; + + std::string version1 = "RVS version"; + std::string version2 = "ROCm version"; + std::string version3 = "amdgpu version"; + std::string OS = "Operating System"; + std::string gpus = "GPUs"; + + // Define column width for consistent spacing + int columnWidth = 14;; + int actionColumnWidth = 32; + int TotalColumnWidth = 69; + int maxGPUNameLength = 24; + + const char boundary = '|'; + + // Function to print a horizontal boundary line + auto printBoundary = [&]() { + std::cout << "+"; + for (int i = 0; i < TotalColumnWidth; ++i) { + std::cout << '-'; + } + std::cout << "+" << std::endl; + }; + + auto printDoubleBoundary = [&]() { + std::cout << "+"; + for (int i = 0; i < TotalColumnWidth; ++i) { + std::cout << '='; + } + std::cout << "+" << std::endl; + }; + + int padding = (TotalColumnWidth - header.size()); + int padleft = padding / 2; + int padright = padding - padleft; + + // Print header row + std::cout << boundary << std::string(padleft, ' ') << header << std::string(padright, ' ') << boundary << std::endl; + + // Print top boundary + printBoundary(); + + std::cout << "\r" << boundary << " " + << std::setw(actionColumnWidth) << std::left << OS + << " | " << std::setw(actionColumnWidth) << std::left << getOSNameVersion() + << " " << boundary << std::endl; + + std::cout << "\r" << boundary << " " + << std::setw(actionColumnWidth) << std::left << version1 + << " | " << std::setw(actionColumnWidth) << std::left << RVS_VERSION_STRING + << " " << boundary << std::endl; + + std::cout << "\r" << boundary << " " + << std::setw(actionColumnWidth) << std::left << version2 + << " | " << std::setw(actionColumnWidth) << std::left << getROCmVersion() + << " " << boundary << std::endl; + + std::cout << "\r" << boundary << " " + << std::setw(actionColumnWidth) << std::left << version3 + << " | " << std::setw(actionColumnWidth) << std::left << getAmdGpuDriverVersion() + << " " << boundary << std::endl; + + rvs::gpulist::Initialize(); + + std::vector gpu_info_list = get_gpu_info(); + + if(gpu_info_list.size()) { + + std::cout << "\r" << boundary << " " + << std::setw(actionColumnWidth) << std::left << "GPUs" + << " | " << std::setw(actionColumnWidth) << std::left << gpu_info_list.size() + << " " << boundary << std::endl; + } + else { + std::cout << "\r" << boundary << " " + << std::setw(actionColumnWidth) << std::left << "GPUs" + << " | " << "\033[31m" << std::setw(actionColumnWidth) << std::left << "No GPUs detected !" << "\033[0m" + << " " << boundary << std::endl; + } + + // Print bottom boundary + printBoundary(); + + std::string GPUdetailsPart1 = "GPU Name - GPU ID"; + std::string GPUdetailsPart2 = "ID - Node ID - BDF"; + padding = (actionColumnWidth - GPUdetailsPart1.size()); + padleft = padding / 2; + padright = padding - padleft; + int gpu_index = 0; + + if(gpu_info_list.size() % 2) { + + std::string GPU1name = gpu_info_list[gpu_index].name; + GPU1name = (GPU1name.size() > maxGPUNameLength) ? GPU1name.substr(0, maxGPUNameLength - 4) + "..." : GPU1name; + + std::string GPU1 = GPU1name + " - " + std::to_string(gpu_info_list[gpu_index].gpu_id); + padright = (actionColumnWidth - GPU1.size()); + + std::cout << "\r" << boundary << " " << std::left << GPUdetailsPart1 << std::string(padding, ' ') << " " << boundary + << " " << GPU1 << std::string(padright, ' ') << " " << boundary + << std::endl; + + GPU1 = std::to_string(gpu_index) + " - " + std::to_string(gpu_info_list[gpu_index].node_id) + " - " + + gpu_info_list[gpu_index].bus; + padright = (actionColumnWidth - GPU1.size()); + + padding = (actionColumnWidth - GPUdetailsPart2.size()); + std::cout << "\r" << boundary << " " << std::left << GPUdetailsPart2 << std::string(padding, ' ') << " " << boundary + << " " << GPU1 << std::string(padright, ' ') << " " << boundary + << std::endl; + + if (1 == gpu_info_list.size()) + printDoubleBoundary(); + else + printBoundary(); + + gpu_index++; + } + else { + + if (gpu_info_list.size()) { + std::cout << "\r" << boundary << " " << std::left << GPUdetailsPart1 << std::string(padding, ' ') << " " << boundary + << " " << std::string(actionColumnWidth, ' ') << " " << boundary + << std::endl; + } + else { + std::string msg = "N/A"; + padright = (actionColumnWidth - msg.size()); + + std::cout << "\r" << boundary << " " << std::left << GPUdetailsPart1 << std::string(padding, ' ') << " " << boundary + << " " << "\033[31m" << msg << std::string(padright, ' ') << " " << "\033[0m" << boundary + << std::endl; + } + + padding = (actionColumnWidth - GPUdetailsPart2.size()); + std::cout << "\r" << boundary << " " << std::left << GPUdetailsPart2 << std::string(padding, ' ') << " " << boundary + << " " << std::string(actionColumnWidth, ' ') << " " << boundary + << std::endl; + + printBoundary(); + } + + for (;gpu_index < gpu_info_list.size();) { + + std::string GPU1name = gpu_info_list[gpu_index].name; + GPU1name = (GPU1name.size() > maxGPUNameLength) ? GPU1name.substr(0, maxGPUNameLength - 4) + "..." : GPU1name; + + std::string GPU1 = GPU1name + " - " + std::to_string(gpu_info_list[gpu_index].gpu_id); + padleft = (actionColumnWidth - GPU1.size()); + + std::string GPU2name = gpu_info_list[gpu_index + 1].name; + GPU2name = (GPU2name.size() > maxGPUNameLength) ? GPU2name.substr(0, maxGPUNameLength - 4) + "..." : GPU2name; + + std::string GPU2 = GPU2name + " - " + std::to_string(gpu_info_list[gpu_index + 1].gpu_id); + padright = (actionColumnWidth - GPU2.size()); + + std::cout << "\r" << boundary + << " " << GPU1 << std::string(padleft, ' ') << " " << boundary + << " " << GPU2 << std::string(padright, ' ') << " " << boundary + << std::endl; + + GPU1 = std::to_string(gpu_index) + " - " + std::to_string(gpu_info_list[gpu_index].node_id) + " - " + + gpu_info_list[gpu_index].bus; + padleft = (actionColumnWidth - GPU1.size()); + + GPU2 = std::to_string(gpu_index + 1) + " - " + std::to_string(gpu_info_list[gpu_index + 1].node_id) + " - " + + gpu_info_list[gpu_index + 1].bus; + padright = (actionColumnWidth - GPU2.size()); + + std::cout << "\r" << boundary + << " " << GPU1 << std::string(padleft, ' ') << " " << boundary + << " " << GPU2 << std::string(padright, ' ') << " " << boundary + << std::endl; + + gpu_index += 2; + + if (gpu_index == gpu_info_list.size()) + printDoubleBoundary(); + else + printBoundary(); + } +} + /** * @brief Executes actions listed in .conf file. * @@ -200,13 +579,117 @@ int rvs::exec::do_yaml(yaml_data_type_t data_type, const std::string& data) { return -1; } + /* Test Summary */ + + const char boundary = '|'; + + // Header columns + std::string header = "ROCm Validation Suite (RVS) Summary"; + std::string header1 = "Action Name"; + std::string header2 = "Module"; + std::string header3 = "Result"; + + // Define column width for consistent spacing + int columnWidth = 14; + int actionColumnWidth = 32; + int TotalColumnWidth = 69; + + // Function to print a horizontal boundary line + auto printBoundary = [&]() { + std::cout << "+"; + for (int i = 0; i < TotalColumnWidth; ++i) { + std::cout << '-'; + } + std::cout << "+" << std::endl; + }; + + // Function to print a horizontal boundary line + auto printDoubleBoundary = [&]() { + std::cout << "+"; + for (int i = 0; i < TotalColumnWidth; ++i) { + std::cout << '='; + } + std::cout << "+" << std::endl; + }; + + int padding = (TotalColumnWidth - header.size()); + int padleft = padding / 2; + int padright = padding - padleft; + + /* Quite logging is enabled */ + if (rvs::options::has_option("-q")) { + + // Print top boundary + printDoubleBoundary(); + + // Print header row + std::cout << boundary << std::string(padleft, ' ') << header << std::string(padright, ' ') << boundary << std::endl; + + // Print top boundary + printDoubleBoundary(); + + systemOverview(); + + // Print header row + std::cout << boundary << " " + << std::setw(actionColumnWidth) << std::left << header1 + << " | " << std::setw(columnWidth) << std::left << header2 + << " | " << std::setw(columnWidth) << std::left << header3 + << " " << boundary << std::endl; + + // Print inner boundary + printDoubleBoundary(); + } + + /* Start of action tests */ + + std::set selected_action_names; + std::set selected_action_indices; + std::string select_val; + if (rvs::options::has_option("-s", &select_val) && !select_val.empty()) { + std::replace(select_val.begin(), select_val.end(), ',', ' '); + std::istringstream iss(select_val); + std::vector tokens; + std::string token; + bool all_numeric = true; + while (iss >> token) { + tokens.push_back(token); + try { + size_t pos = 0; + int idx = std::stoi(token, &pos); + if (pos != token.size() || idx < 0) + all_numeric = false; + } catch (...) { + all_numeric = false; + } + } + for (const auto& t : tokens) { + if (all_numeric) + selected_action_indices.insert(std::stoi(t)); + else + selected_action_names.insert(t); + } + } + bool has_selection = !selected_action_names.empty() || + !selected_action_indices.empty(); + /* Number of times to execute the test */ for (int i = 0; i < num_times; i++) { + int action_idx = 0; // for all actions... - for (YAML::const_iterator it = actions.begin(); it != actions.end(); ++it) { + for (YAML::const_iterator it = actions.begin(); it != actions.end(); + ++it, ++action_idx) { const YAML::Node& action = *it; + if (has_selection) { + std::string action_name = action["name"].as(); + if (selected_action_names.find(action_name) == selected_action_names.end() && + selected_action_indices.find(action_idx) == selected_action_indices.end()) { + continue; + } + } + sts = 0; rvs::logger::log("Action name :" + action["name"].as(), rvs::logresults); @@ -287,12 +770,49 @@ int rvs::exec::do_yaml(yaml_data_type_t data_type, const std::string& data) { pif1->callback_set(&rvs::exec::action_callback, (void *)this); } + exec_action action_info; + + action_info.name = action["name"].as(); + action_info.module = action["module"].as(); + + std::transform(action_info.module.begin(), + action_info.module.end(), + action_info.module.begin(), ::toupper); + + std::thread in_progress_t; + + if (rvs::options::has_option("-q")) { + + in_progress = true; + + // Start compute workload thread + in_progress_t = std::thread(&rvs::exec::in_progress_thread, this, action_info); + + } + // execute action sts = pif1->run(); + if (rvs::options::has_option("-q")) { + + in_progress = false; + + in_progress_t.join(); + + std::string actionresult = (!sts) ? "PASS" : "FAIL"; + std::string textcolor = (!sts) ? "\033[32m" : "\033[31m"; + + std::cout << "\r" << boundary << " " + << std::setw(actionColumnWidth) << std::left << action_info.name + << " | " << std::setw(columnWidth) << std::left << action_info.module + << " | " << textcolor << std::setw(columnWidth) << std::left << actionresult << "\033[0m" + << " " << boundary << std::endl; + } + // processing finished, release action object module::action_destroy(pa); + // Action pass fail status !! // errors? if (sts) { @@ -303,18 +823,107 @@ int rvs::exec::do_yaml(yaml_data_type_t data_type, const std::string& data) { result.output_log = buff; callback(&result); - rvs::logger::Err("Action failed to run successfully.", - action["module"].as().c_str(), - action["name"].as().c_str()); + if (!rvs::options::has_option("-q")) { + rvs::logger::Err("Action failed to run successfully.", + action["module"].as().c_str(), + action["name"].as().c_str()); + } + + action_info.result = false; + } + else { + action_info.result = true; + } + action_details.push_back(action_info); + } + } + /* End of action tests */ + + /* Quite logging is not enabled */ + if (!rvs::options::has_option("-q")) { + + const char boundary = '|'; + + // Header columns + std::string header = "ROCm Validation Suite (RVS) Summary"; + std::string header1 = "Action Name"; + std::string header2 = "Module"; + std::string header3 = "Result"; + + // Define column width for consistent spacing + int columnWidth = 14; + int actionColumnWidth = 32; + int TotalColumnWidth = 69; + + // Function to print a horizontal boundary line + auto printBoundary = [&]() { + std::cout << "+"; + for (int i = 0; i < TotalColumnWidth; ++i) { + std::cout << '-'; } + std::cout << "+" << std::endl; + }; + + // Function to print a horizontal boundary line + auto printDoubleBoundary = [&]() { + std::cout << "+"; + for (int i = 0; i < TotalColumnWidth; ++i) { + std::cout << '='; + } + std::cout << "+" << std::endl; + }; + + int padding = (TotalColumnWidth - header.size()); + int padleft = padding / 2; + int padright = padding - padleft; + + // Print top boundary + printDoubleBoundary(); + + // Print header row + std::cout << boundary << std::string(padleft, ' ') << header << std::string(padright, ' ') << boundary << std::endl; + + // Print top boundary + printDoubleBoundary(); + + systemOverview(); + + // Print header row + std::cout << boundary << " " + << std::setw(actionColumnWidth) << std::left << header1 + << " | " << std::setw(columnWidth) << std::left << header2 + << " | " << std::setw(columnWidth) << std::left << header3 + << " " << boundary << std::endl; + + // Print inner boundary + printDoubleBoundary(); + + for(size_t i = 0; i < action_details.size(); i++) { + std::string result = (action_details[i].result) ? "PASS" : "FAIL"; + std::string textcolor = (action_details[i].result) ? "\033[32m" : "\033[31m"; + + // Print data row + std::cout << boundary << " " + << std::setw(actionColumnWidth) << std::left << action_details[i].name + << " | " << std::setw(columnWidth) << std::left << action_details[i].module + << " | " << textcolor << std::setw(columnWidth) << std::left << result << "\033[0m" + << " " << boundary << std::endl; } + // Print bottom boundary + printBoundary(); + } + else { + printBoundary(); + } + if (rvs::logger::to_json()) { + rvs::lp::JsonEndNodeCreate(); } - result.status = RVS_STATUS_SUCCESS; result.output_log = "RVS session successfully completed."; callback(&result); + return 0; } @@ -395,6 +1004,12 @@ int rvs::exec::do_yaml_properties(const YAML::Node& node, } } + string duration; + if (rvs::options::has_option("-t", &duration)) { + uint64_t dur_ms = std::stoull(duration) * 1000; + sts += pif1->property_set("duration", std::to_string(dur_ms)); + } + return sts; } @@ -437,16 +1052,9 @@ bool rvs::exec::is_yaml_properties_collection( if (property_name == "io_links-properties") return true; - } else { - if (module_name == "peqt") { - if (property_name == "capability") { - return true; - } - } else { - if (module_name == "gm") { - if (property_name == "metrics") - return true; - } + } else if (module_name == "peqt") { + if (property_name == "capability") { + return true; } } diff --git a/rvs/src/rvsmodule.cpp b/rvs/src/rvsmodule.cpp index 3d4bf1644..d84db742f 100644 --- a/rvs/src/rvsmodule.cpp +++ b/rvs/src/rvsmodule.cpp @@ -40,12 +40,54 @@ #include "include/rvsaction.h" #include "include/rvsliblog.h" #include "include/rvsoptions.h" -#ifdef FETCH_ROCMPATH_FROM_ROCMCORE -#include "rocm-core/rocm_getpath.h" -#endif +#include "include/rvs_util.h" #define MODULE_NAME_CAPS "CLI" +namespace { + +void rvs_prefer_load_err(std::string* keep, const std::string& next) { + if (next.empty()) { + return; + } + if (keep->empty() || + (keep->find("could not resolve module path") != std::string::npos && + next.find("could not resolve module path") == std::string::npos)) { + *keep = next; + } +} + +/** + * @brief dlopen a module .so after rvs_verify_module_so_for_dlopen(). + * + * Runs verification first; on success opens the canonical path with RTLD_NOW. + * Returns nullptr and sets err_msg when verification or dlopen fails. Used for + * all four module search paths in find_create_module(). + * + * @param so_path Candidate .so path. + * @param err_msg Optional; verification or dlopen failure reason. + * @return dlopen handle, or nullptr on failure. + */ +void* rvs_dlopen_verified_module(const std::string& so_path, + std::string* err_msg) { + std::string canonical; + std::string verify_err; + if (!rvs_verify_module_so_for_dlopen(so_path, &verify_err, &canonical)) { + if (err_msg) { + *err_msg = verify_err; + } + return nullptr; + } + void* handle = dlopen(canonical.c_str(), RTLD_NOW); + if (!handle && err_msg) { + const char* dl_err = dlerror(); + *err_msg = dl_err ? dl_err : "dlopen failed"; + } + return handle; +} + +} // namespace + std::map rvs::module::modulemap; std::map rvs::module::filemap; YAML::Node rvs::module::config; @@ -164,60 +206,56 @@ rvs::module* rvs::module::find_create_module(const char* name) { // open module .so library string libpath; - rvs::options::has_option("pwd", &libpath); // has ending forward slash too - libpath += "../lib/rvs/"; + std::string load_err; + void* psolib = nullptr; - string sofullname(libpath + it->second); - void* psolib = dlopen(sofullname.c_str(), RTLD_NOW); - // error? + // Dev build layout: modules live next to rvs in build/bin/ + if (rvs::options::has_option("pwd", &libpath)) { + string sofullname(libpath + it->second); + psolib = rvs_dlopen_verified_module(sofullname, &load_err); + } + if (!psolib) { + // Install layout: prefix/lib/rvs/ + if (rvs::options::has_option("pwd", &libpath)) { + libpath += "../lib/rvs/"; + } else { + libpath = "../lib/rvs/"; + } + string sofullname(libpath + it->second); + std::string try_err; + psolib = rvs_dlopen_verified_module(sofullname, &try_err); + rvs_prefer_load_err(&load_err, try_err); + } if (!psolib) { - //Search libraries in current path set in pwd option for backward compatibility - if(false == rvs::options::has_option("pwd", &libpath)) { - //Search libraries in current path if pwd option not set + // pwd-relative fallback for backward compatibility + if (rvs::options::has_option("pwd", &libpath)) { + // libpath reset to pwd by has_option + } else { libpath = "./"; - } // has ending forward slash too + } string sofullname(libpath + it->second); - psolib = dlopen(sofullname.c_str(), RTLD_NOW); - // error? + std::string try_err; + psolib = rvs_dlopen_verified_module(sofullname, &try_err); + rvs_prefer_load_err(&load_err, try_err); + } + if (!psolib) { + // Final fallback: install prefix lib dir from rvs binary path or RVS_LIB_PATH + libpath = rvs_get_rvs_modules_lib_dir_string(); + libpath += "/"; + string sofullname(libpath + it->second); + std::string try_err; + psolib = rvs_dlopen_verified_module(sofullname, &try_err); + rvs_prefer_load_err(&load_err, try_err); if (!psolib) { - //Search libraries in RVS install path -#ifdef FETCH_ROCMPATH_FROM_ROCMCORE - char *installPath = nullptr; - unsigned int installPathLen = 0; - string rocmPath; - PathErrors_t retVal = PathSuccess; - // Get the ROCm install path - retVal = getROCmInstallPath( &installPath, &installPathLen ); - if(retVal == PathSuccess){ - rocmPath = installPath; - } - else { - std::cout << "Failed to get ROCm Install Path: " << retVal <<"\nSet ROCM_PATH in env" << std::endl; - } - // free allocated memory - if(installPath != nullptr) { - free(installPath); - } - libpath = rocmPath; - libpath += "/"; - libpath += RVS_LIB_PATH; -#else - libpath = RVS_LIB_PATH; -#endif - libpath += "/"; - string sofullname(libpath + it->second); - psolib = dlopen(sofullname.c_str(), RTLD_NOW); - if (!psolib) { char buff[1024]; snprintf(buff, sizeof(buff), "could not load .so '%s'", sofullname.c_str()); rvs::logger::Err(buff, MODULE_NAME_CAPS); snprintf(buff, sizeof(buff), - "reason: '%s'", dlerror()); + "reason: '%s'", load_err.c_str()); rvs::logger::Err(buff, MODULE_NAME_CAPS); return NULL; // fail } - } } // create module object m = new rvs::module(name, psolib); @@ -273,6 +311,7 @@ int rvs::module::initialize() { d.cbLogExt = rvs::logger::LogExt; d.cbLogRecordCreate = rvs::logger::LogRecordCreate; d.cbJsonNamedListCreate = rvs::logger::JsonNamedListCreate; + d.cbJsonNestedListCreate = rvs::logger::JsonNestedListCreate; d.cbJsonStartNodeCreate = rvs::logger::JsonStartNodeCreate; d.cbJsonActionStartNodeCreate = rvs::logger::JsonActionStartNodeCreate; d.cbJsonEndNodeCreate = rvs::logger::JsonEndNodeCreate; diff --git a/rvs/tests.cmake b/rvs/tests.cmake index dbfa79cd0..6c9041e2a 100644 --- a/rvs/tests.cmake +++ b/rvs/tests.cmake @@ -36,14 +36,12 @@ set(HIPBLASLT_LIB "hipblaslt") set(CORE_RUNTIME_NAME "hsa-runtime") set(CORE_RUNTIME_TARGET "${CORE_RUNTIME_NAME}64") -find_package(OpenMP) - ## define lib directories link_directories(${RVS_LIB_DIR} ${ROCBLAS_LIB_DIR} ${ROCM_SMI_LIB_DIR} ${HIPRAND_LIB_DIR}) ## define target for "test-to-fail" add_executable(${RVS_TARGET}fail src/rvs.cpp) -target_link_libraries(${RVS_TARGET}fail rvslib rvslibut ${PROJECT_LINK_LIBS} OpenMP::OpenMP_CXX +target_link_libraries(${RVS_TARGET}fail rvslib rvslibut ${PROJECT_LINK_LIBS} -fopenmp ${ROCM_SMI_LIB} ${ROCBLAS_LIB} ${ROCM_CORE} ${CORE_RUNTIME_TARGET} ${HIPRAND_LIB} ${HIPBLASLT_LIB}) target_compile_definitions(${RVS_TARGET}fail PRIVATE RVS_INVERT_RETURN_STATUS) @@ -212,7 +210,7 @@ FOREACH(SINGLE_TEST ${TESTSOURCES}) target_link_libraries(${TEST_NAME} ${PROJECT_LINK_LIBS} ${PROJECT_TEST_LINK_LIBS} - rvslib rvslibut gtest_main gtest pthread OpenMP::OpenMP_CXX + rvslib rvslibut gtest_main gtest pthread -fopenmp ${ROCM_SMI_LIB} ${ROCBLAS_LIB} ${CORE_RUNTIME_TARGET} ${ROCM_CORE} ${HIPRAND_LIB} ${HIPBLASLT_LIB} ) add_dependencies(${TEST_NAME} rvs_gtest_target) diff --git a/rvs/wrongvernum.config b/rvs/wrongvernum.config index c7b11867d..bfd6ce70b 100644 --- a/rvs/wrongvernum.config +++ b/rvs/wrongvernum.config @@ -1,14 +1,10 @@ version: 1 gpup: libgpup.so peqt: libpeqt.so -pesm: libpesm.so rcqt: librcqt.so -smqt: libsmqt.so -gm: libgm.so gst: libgst.so pbqt: libpbqt.so pebb: libpebb.so iet: libiet.so mem: libmem.so babel: libbabel.so -edp: libedp.so diff --git a/rvs_nightly_docker.sh b/rvs_nightly_docker.sh new file mode 100755 index 000000000..b38eeb96f --- /dev/null +++ b/rvs_nightly_docker.sh @@ -0,0 +1,871 @@ +#!/usr/bin/env bash +################################################################################ +# RVS nightly tests inside a ROCm-matched docker container. +# +# Default (RVS_DOCKER_ON_TARGET=true): build the ROCm image on the GPU target +# (RVS_DOCKER_BUILD_ON_TARGET=true) or deliver via scp/registry, then docker run via SSH. +# Set RVS_DOCKER_ON_TARGET=false to run docker locally on the build host instead. +# +# Image delivery (RVS_DOCKER_TRANSFER_MODE): +# auto build-on-target if RVS_DOCKER_BUILD_ON_TARGET=true, else registry +# when RVS_DOCKER_REGISTRY is set, else scp (default path) +# scp docker save | (pigz|gzip) | scp | docker load +# registry docker push on build host, docker pull on target +# build-on-target docker build on the GPU node (no cross-host image transfer) +# +# Set RVS_DOCKER_SKIP_IF_PRESENT=false to force re-transfer/re-build. +################################################################################ + +set -euo pipefail + +DOCKER_IMAGE="${RVS_NIGHTLY_DOCKER_IMAGE:-rvs-nightly-rocm:latest}" +DOCKER_IMAGE_ARCHIVE="${RVS_DOCKER_IMAGE_ARCHIVE:-rvs-nightly-rocm-image.tar.gz}" +REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +DOCKER_BUILD_DIR="${RVS_DOCKER_BUILD_DIR:-${REPO_ROOT}/.github/docker/rvs-nightly-rocm}" +if [[ "$DOCKER_BUILD_DIR" != /* ]]; then + DOCKER_BUILD_DIR="${REPO_ROOT}/${DOCKER_BUILD_DIR#./}" +fi + +usage() { + cat <<'EOF' +Usage: rvs_nightly_docker.sh + +Image delivery (build host → target): + ensure-image-on-target Skip, registry pull, scp load, or build on target (preferred) + transfer-image-to-target Alias for ensure-image-on-target + build-image-on-target tar docker build context, scp to target, docker build + verify-image-on-target Confirm image exists on target + +In-container steps (on target when RVS_DOCKER_ON_TARGET=true): + pull-image Verify image on build host (local) before scp/registry push + verify-rocm rocminfo + amd-smi inside container + install-rvs Extract RVS tarball under /opt/rocm/extras- + run-level4 rvs -r 4 inside container + capture-versions Write rvs_version and target_rocm_version to GITHUB_OUTPUT + run-pipeline verify-rocm → install-rvs → run-level4 + +Environment: + RVS_DOCKER_TRANSFER_MODE auto | scp | registry | build-on-target (default: auto) + RVS_DOCKER_BUILD_ON_TARGET true (default) prefers build-on-target in auto mode + RVS_DOCKER_REGISTRY e.g. ghcr.io/org/rvs-nightly-rocm (enables registry mode) + RVS_DOCKER_REGISTRY_USER optional registry login user + RVS_DOCKER_REGISTRY_PASSWORD optional registry login password + RVS_DOCKER_ROCM_VERSION expected ROCm SDK version (for skip-if-present matching) + RVS_DOCKER_SKIP_IF_PRESENT true (default) skips transfer when target already has image + RVS_DOCKER_BUILD_DIR docker build context (default: .github/docker/rvs-nightly-rocm) + RVS_DOCKER_SDK_FALLBACK_LATEST when true (default), fall back to latest same-line, same-major, then newest nightly SDK + ROCM_SDK_NIGHTLY_BASE_URL optional ROCm SDK nightly download host (vars.ROCM_SDK_NIGHTLY_BASE_URL) + ROCM_SDK_NIGHTLY_INDEX_URL optional listing URL (defaults to BASE_URL/ when unset) +EOF +} + +docker_build_fallback_args() { + case "${RVS_DOCKER_SDK_FALLBACK_LATEST:-true}" in + true|1|yes|YES) printf '%s' ' --fallback-latest-sdk' ;; + *) printf '%s' '' ;; + esac +} + +require_env() { + local n="$1" + if [ -z "${!n:-}" ]; then + echo "::error::Required environment variable $n is not set" >&2 + exit 1 + fi +} + +use_target_docker() { + case "${RVS_DOCKER_ON_TARGET:-true}" in + 0|false|False|no|NO) return 1 ;; + esac + if [ -n "${TARGET_NODE:-}" ] && [ "$TARGET_NODE" != "localhost" ] && [ "$TARGET_NODE" != "127.0.0.1" ]; then + return 0 + fi + if [ -z "${TARGET_NODE:-}" ] || [ "$TARGET_NODE" = "localhost" ] || [ "$TARGET_NODE" = "127.0.0.1" ]; then + if [ "${RVS_DOCKER_ON_TARGET:-true}" = "true" ] || [ "${RVS_DOCKER_ON_TARGET:-}" = "1" ]; then + echo "::error::RVS_DOCKER_ON_TARGET is enabled but TARGET_NODE is not set (use secrets.RVS_TARGET_NODE)." >&2 + exit 1 + fi + fi + return 1 +} + +rvs_in_image() { + case "${RVS_DOCKER_RVS_IN_IMAGE:-false}" in + 1|true|True|yes|YES) return 0 ;; + esac + return 1 +} + +container_rocm_path() { + printf '%s\n' "${TARGET_ROCM_PATH:-/opt/rocm/install}" +} + +require_ssh_config() { + if [ -z "${SSH_CONFIG_FILE:-}" ]; then + local ssh_env_file="${RUNNER_TEMP:-/tmp}/rvs_ssh_env.sh" + if [ -f "$ssh_env_file" ]; then + # shellcheck source=/dev/null + source "$ssh_env_file" + fi + fi + require_env SSH_CONFIG_FILE + if [ ! -f "$SSH_CONFIG_FILE" ]; then + echo "::error::SSH config missing at $SSH_CONFIG_FILE — run: ./rvs_nightly_test.sh setup-ssh" >&2 + exit 1 + fi +} + +phase_start() { + PHASE_LABEL="$1" + PHASE_START_TS=$(date +%s) + echo "::group::${PHASE_LABEL}" + echo "[$(date -u +%FT%TZ)] START ${PHASE_LABEL}" +} + +phase_end() { + local elapsed=$(( $(date +%s) - PHASE_START_TS )) + echo "[$(date -u +%FT%TZ)] END ${PHASE_LABEL} (${elapsed}s)" + echo "::notice::${PHASE_LABEL} completed in ${elapsed}s" + echo "::endgroup::" +} + +compress_pipe() { + if command -v pigz >/dev/null 2>&1; then + echo "::notice::Compressing with pigz" + pigz -1 + else + echo "::notice::Compressing with gzip (install pigz for faster compression)" + gzip -1 + fi +} + +decompress_cmd() { + if command -v pigz >/dev/null 2>&1; then + pigz -dc + else + gzip -dc + fi +} + +rocm_version_from_tarball() { + local base="${1##*/}" + if [[ "$base" =~ -r([0-9]{2})([0-9]{2})\.([0-9]{8})-Linux\.tar\.gz$ ]]; then + printf '%d.%d.0a%s\n' "$((10#${BASH_REMATCH[1]}))" "$((10#${BASH_REMATCH[2]}))" "${BASH_REMATCH[3]}" + fi +} + +expected_rocm_version() { + if [ -n "${RVS_DOCKER_ROCM_VERSION:-}" ]; then + printf '%s\n' "$RVS_DOCKER_ROCM_VERSION" + return 0 + fi + if [ -n "${TARBALL_NAME:-}" ]; then + rocm_version_from_tarball "$TARBALL_NAME" + return 0 + fi + return 1 +} + +image_rocm_version_on_target() { + require_ssh_config + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "docker image inspect '${DOCKER_IMAGE}' --format '{{range .Config.Env}}{{println .}}{{end}}' 2>/dev/null" \ + | sed -n 's/^ROCM_VERSION=//p' | head -1 +} + +image_rocm_version_local() { + docker image inspect "${DOCKER_IMAGE}" --format '{{range .Config.Env}}{{println .}}{{end}}' 2>/dev/null \ + | sed -n 's/^ROCM_VERSION=//p' | head -1 +} + +build_host_image_matches_expected() { + local expected actual + expected="$(expected_rocm_version 2>/dev/null || true)" + if [ -z "$expected" ]; then + return 0 + fi + actual="$(image_rocm_version_local 2>/dev/null || true)" + [ "$actual" = "$expected" ] +} + +verify_gzip_archive() { + local archive="$1" + local label="${2:-archive}" + if [ ! -s "$archive" ]; then + echo "::error::${label} is empty: ${archive}" >&2 + exit 1 + fi + if command -v pigz >/dev/null 2>&1; then + pigz -t "$archive" + else + gzip -t "$archive" + fi + echo "::notice::${label} integrity OK ($(stat -c%s "$archive" 2>/dev/null || stat -f%z "$archive") bytes)" +} + +verify_remote_file_size() { + local local_path="$1" + local remote_path="$2" + local local_size remote_size + local_size="$(stat -c%s "$local_path" 2>/dev/null || stat -f%z "$local_path")" + remote_size="$(ssh -q -F "$SSH_CONFIG_FILE" rvs-target "stat -c%s '${remote_path}' 2>/dev/null || stat -f%z '${remote_path}'")" + if [ "$local_size" != "$remote_size" ]; then + echo "::error::Remote file size mismatch for ${remote_path}: local=${local_size} remote=${remote_size}" >&2 + exit 1 + fi + echo "::notice::Remote file size matches local (${local_size} bytes)" +} + +cmd_build_image_on_build_host() { + if ! rvs_in_image; then + require_env TARBALL_NAME + fi + check_docker_local + local build_script="${REPO_ROOT}/.github/docker/build-rocm-sdk-image.sh" + if [ ! -f "$build_script" ]; then + echo "::error::Docker build script not found: ${build_script}" >&2 + exit 1 + fi + chmod +x "$build_script" + local fallback_args=() + case "${RVS_DOCKER_SDK_FALLBACK_LATEST:-true}" in + true|1|yes|YES) fallback_args=(--fallback-latest-sdk) ;; + esac + phase_start "docker build on build host" + if [ -n "${TARBALL_NAME:-}" ]; then + RVS_NIGHTLY_DOCKER_IMAGE="${DOCKER_IMAGE}" "$build_script" --context "${DOCKER_BUILD_DIR}" --from-tarball "${TARBALL_NAME}" "${fallback_args[@]}" + else + RVS_NIGHTLY_DOCKER_IMAGE="${DOCKER_IMAGE}" "$build_script" --context "${DOCKER_BUILD_DIR}" --channel nightly "${fallback_args[@]}" + fi + phase_end "docker build on build host" + if ! build_host_image_matches_expected; then + echo "::warning::Build host image ROCM_VERSION=$(image_rocm_version_local || echo unknown) may differ from expected $(expected_rocm_version 2>/dev/null || echo unknown) (SDK fallback in use?)" + fi +} + +ensure_build_host_image() { + check_docker_local + if docker image inspect "$DOCKER_IMAGE" >/dev/null 2>&1 && build_host_image_matches_expected; then + echo "::notice::Build host has ${DOCKER_IMAGE} with matching ROCM_VERSION" + return 0 + fi + if [ -z "${TARBALL_NAME:-}" ] && ! rvs_in_image; then + echo "::error::Build host image is missing or stale and TARBALL_NAME is unset — cannot rebuild." >&2 + echo "Re-run with build_docker_image=true, build_on_target=true, or set TARBALL_NAME." >&2 + exit 1 + fi + local expected actual + expected="$(expected_rocm_version 2>/dev/null || true)" + actual="$(image_rocm_version_local 2>/dev/null || true)" + echo "::notice::Build host image ROCM_VERSION=${actual:-missing}; need ${expected:-unknown} — rebuilding on build host" + cmd_build_image_on_build_host +} + +resolve_transfer_mode() { + case "${RVS_DOCKER_TRANSFER_MODE:-auto}" in + scp|registry|build-on-target) + printf '%s\n' "${RVS_DOCKER_TRANSFER_MODE}" + ;; + auto) + case "${RVS_DOCKER_BUILD_ON_TARGET:-true}" in + 0|false|False|no|NO) + if [ -n "${RVS_DOCKER_REGISTRY:-}" ]; then + printf '%s\n' registry + elif [ -n "${TARBALL_NAME:-}" ]; then + # Prefer build-on-target over scp when the tarball is known — scp of + # multi-GB images is slow and fragile; build-on-target sends only context. + printf '%s\n' build-on-target + else + printf '%s\n' scp + fi + ;; + *) + printf '%s\n' build-on-target + ;; + esac + ;; + *) + echo "::error::Invalid RVS_DOCKER_TRANSFER_MODE: ${RVS_DOCKER_TRANSFER_MODE}" >&2 + exit 1 + ;; + esac +} + +registry_image_ref() { + local version="${1:-}" + local base="${RVS_DOCKER_REGISTRY%/}" + if [ -n "$version" ]; then + printf '%s:%s\n' "$base" "$version" + else + printf '%s\n' "$base" + fi +} + +docker_registry_login() { + local where="$1" + if [ -z "${RVS_DOCKER_REGISTRY_USER:-}" ] || [ -z "${RVS_DOCKER_REGISTRY_PASSWORD:-}" ]; then + echo "::notice::No registry credentials set — assuming public pull/push for ${where}" + return 0 + fi + local registry_host="${RVS_DOCKER_REGISTRY%%/*}" + case "$where" in + local) + echo "${RVS_DOCKER_REGISTRY_PASSWORD}" | docker login -u "${RVS_DOCKER_REGISTRY_USER}" \ + --password-stdin "$registry_host" + ;; + target) + ssh -q -F "$SSH_CONFIG_FILE" rvs-target bash -s <&2 + exit 1 + ;; + esac +} + +skip_if_present_enabled() { + case "${RVS_DOCKER_SKIP_IF_PRESENT:-true}" in + 0|false|False|no|NO) return 1 ;; + esac + return 0 +} + +image_ready_on_target() { + require_ssh_config + check_docker_on_target + if ! ssh -q -F "$SSH_CONFIG_FILE" rvs-target "docker image inspect '${DOCKER_IMAGE}' >/dev/null 2>&1"; then + return 1 + fi + local expected actual + expected="$(expected_rocm_version 2>/dev/null || true)" + if [ -z "$expected" ]; then + return 0 + fi + actual="$(image_rocm_version_on_target || true)" + if [ "$actual" = "$expected" ]; then + echo "::notice::Target already has ${DOCKER_IMAGE} (ROCM_VERSION=${expected}) — skipping delivery" + return 0 + fi + echo "Target image ROCM_VERSION=${actual:-unknown}; need ${expected} — delivery required" + return 1 +} + +target_rvs_install_dir() { + require_env REMOTE_WORK_DIR + require_env ROCM_MAJOR + printf '%s/extras-%s' "${REMOTE_WORK_DIR%/}" "$ROCM_MAJOR" +} + +docker_gpu_opts() { + echo --ipc=host --network=host \ + --device=/dev/kfd \ + --device=/dev/dri \ + --group-add video \ + --cap-add=SYS_PTRACE \ + --security-opt seccomp=unconfined +} + +check_docker_local() { + if docker info >/dev/null 2>&1; then + return 0 + fi + echo "::error::Cannot access Docker on the build host (/var/run/docker.sock). Add the runner user to the 'docker' group and restart the runner." >&2 + exit 1 +} + +check_docker_on_target() { + require_ssh_config + if ssh -q -F "$SSH_CONFIG_FILE" rvs-target 'docker info >/dev/null 2>&1'; then + return 0 + fi + echo "::error::Cannot access Docker on the target node. Ensure the SSH user is in the 'docker' group on the target." >&2 + exit 1 +} + +docker_run_local() { + check_docker_local + local -a install_mount=() + local rocm_path + rocm_path="$(container_rocm_path)" + if ! rvs_in_image && [ -n "${INSTALL_DIR:-}" ] && [ -n "${REMOTE_WORK_DIR:-}" ] && [ -n "${ROCM_MAJOR:-}" ]; then + local host_install + host_install="$(target_rvs_install_dir)" + mkdir -p "$host_install" + install_mount=(-v "${host_install}:${INSTALL_DIR}") + fi + if ! rvs_in_image && [ -n "${TARBALL_NAME:-}" ] && [ -f "${REPO_ROOT}/pkg/${TARBALL_NAME}" ]; then + install_mount+=(-v "${REPO_ROOT}/pkg/${TARBALL_NAME}:/pkg/${TARBALL_NAME}:ro") + fi + # shellcheck disable=SC2046 + docker run --rm \ + $(docker_gpu_opts) \ + "${install_mount[@]}" \ + -v "${REPO_ROOT}/reports:/reports" \ + -w /workspace \ + -e "ROCM_PATH=${rocm_path}" \ + -e "TARGET_ROCM_PATH=${rocm_path}" \ + "$DOCKER_IMAGE" \ + bash -lc "$1" +} + +target_docker_run() { + require_ssh_config + require_env REMOTE_WORK_DIR + check_docker_on_target + local inner_cmd="$1" + local gpu_opts install_vol pkg_path pkg_mount remote_dirs escaped + gpu_opts=$(docker_gpu_opts) + install_vol="" + pkg_path="" + pkg_mount="" + remote_dirs="'${REMOTE_WORK_DIR}/reports' '${REMOTE_WORK_DIR}/workspace' '${REMOTE_WORK_DIR}/pkg'" + local rocm_path + rocm_path="$(container_rocm_path)" + if ! rvs_in_image && [ -n "${INSTALL_DIR:-}" ] && [ -n "${ROCM_MAJOR:-}" ]; then + install_vol="-v ${REMOTE_WORK_DIR}/extras-${ROCM_MAJOR}:${INSTALL_DIR}" + # Pre-create so docker does not make a root-owned extras dir. + remote_dirs="${remote_dirs} '${REMOTE_WORK_DIR}/extras-${ROCM_MAJOR}'" + fi + if ! rvs_in_image && [ -n "${TARBALL_NAME:-}" ]; then + pkg_path="${REMOTE_WORK_DIR}/pkg/${TARBALL_NAME}" + pkg_mount="-v ${pkg_path}:/pkg/${TARBALL_NAME}:ro" + fi + escaped=$(printf '%q' "$inner_cmd") + # Only bind-mount the tarball when the file already exists on the target. + # If the path is missing, docker would create a root-owned *directory* there, + # and a later scp to that path fails with Permission denied. + ssh -q -F "$SSH_CONFIG_FILE" rvs-target bash -s <&2 + rm -rf "${pack_dir}" + exit 1 + fi + tar -C "${pack_dir}" -czf "${local_archive}" . + rm -rf "${pack_dir}" + ls -lh "${local_archive}" + ssh -q -F "$SSH_CONFIG_FILE" rvs-target "mkdir -p '${remote_build_dir}'" + scp -q -F "$SSH_CONFIG_FILE" "${local_archive}" \ + "rvs-target:${remote_archive}" + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "tar -xzf '${remote_archive}' -C '${remote_build_dir}' && rm -f '${remote_archive}'" + phase_end "Sync docker build context to target" + + phase_start "docker build on GPU target" + local fallback_args build_args="" + fallback_args="$(docker_build_fallback_args)" + if [ -n "${TARBALL_NAME:-}" ]; then + build_args="--from-tarball '${TARBALL_NAME}'${fallback_args}" + else + build_args="--channel nightly${fallback_args}" + fi + ssh -q -F "$SSH_CONFIG_FILE" rvs-target bash -s < "${archive}" + ls -lh "${archive}" + verify_gzip_archive "${archive}" "local image archive" + phase_end "docker save + compress (${DOCKER_IMAGE})" + + phase_start "scp image archive to target" + ssh -q -F "$SSH_CONFIG_FILE" rvs-target "mkdir -p '${REMOTE_WORK_DIR}/docker'" + ssh -q -F "$SSH_CONFIG_FILE" rvs-target "rm -f '${remote_archive}' '${remote_archive}.partial'" + scp -q -F "$SSH_CONFIG_FILE" "${archive}" \ + "rvs-target:${remote_archive}.partial" + ssh -q -F "$SSH_CONFIG_FILE" rvs-target "mv -f '${remote_archive}.partial' '${remote_archive}'" + verify_remote_file_size "${archive}" "${remote_archive}" + ssh -q -F "$SSH_CONFIG_FILE" rvs-target bash -s </dev/null 2>&1; then + pigz -t '${remote_archive}' +else + gzip -t '${remote_archive}' +fi +REMOTE + phase_end "scp image archive to target" + + phase_start "docker load on target" + local decompress + decompress="$(decompress_cmd)" + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "${decompress} '${remote_archive}' | docker load" + phase_end "docker load on target" + cmd_verify_image_on_target +} + +cmd_download_rvs_tarball() { + require_env TARBALL_URL + require_env TARBALL_NAME + require_ssh_config + require_env REMOTE_WORK_DIR + + local local_pkg="${REPO_ROOT}/pkg/${TARBALL_NAME}" + mkdir -p "${REPO_ROOT}/pkg" + + phase_start "Download RVS tarball on orchestrator" + echo " ${TARBALL_URL}" + curl -fL --max-time 600 -o "${local_pkg}" "${TARBALL_URL}" + file "${local_pkg}" || true + ls -lh "${local_pkg}" + phase_end "Download RVS tarball on orchestrator" + + phase_start "scp RVS tarball to target" + echo " local_pkg=${local_pkg}" + echo " REMOTE_WORK_DIR=${REMOTE_WORK_DIR}" + echo " remote_dest=${REMOTE_WORK_DIR}/pkg/${TARBALL_NAME}" + # Clear a leftover root-owned dir/file from a prior docker bind-mount miss, + # then stage the tarball as a normal user-owned file. + ssh -q -F "$SSH_CONFIG_FILE" rvs-target bash -s </dev/null || sudo -n rm -rf "\$dest" +fi +REMOTE + scp -q -F "$SSH_CONFIG_FILE" "${local_pkg}" \ + "rvs-target:${REMOTE_WORK_DIR}/pkg/${TARBALL_NAME}" + phase_end "scp RVS tarball to target" +} + +cmd_ensure_image_on_target() { + require_ssh_config + require_env REMOTE_WORK_DIR + + local mode requested + requested="${RVS_DOCKER_TRANSFER_MODE:-auto}" + mode="$(resolve_transfer_mode)" + if [ "$requested" = "auto" ] && [ "$mode" = "build-on-target" ] && \ + case "${RVS_DOCKER_BUILD_ON_TARGET:-true}" in 0|false|False|no|NO) true ;; *) false ;; esac; then + echo "::notice::auto mode: using build-on-target (TARBALL_NAME set; avoids fragile multi-GB scp). Set RVS_DOCKER_TRANSFER_MODE=scp to force scp." + fi + echo "Image delivery mode: ${mode}" + + if skip_if_present_enabled && image_ready_on_target; then + return 0 + fi + + case "$mode" in + build-on-target) + if [ -z "${TARBALL_NAME:-}" ] && ! rvs_in_image; then + echo "::error::build-on-target requires TARBALL_NAME" >&2 + exit 1 + fi + cmd_build_image_on_target + ;; + registry) + cmd_transfer_via_registry + ;; + scp) + cmd_transfer_via_scp + ;; + esac +} + +cmd_verify_image_on_target() { + require_ssh_config + check_docker_on_target + if ! ssh -q -F "$SSH_CONFIG_FILE" rvs-target "docker image inspect '${DOCKER_IMAGE}' >/dev/null 2>&1"; then + echo "::error::Image ${DOCKER_IMAGE} not found on target after delivery." >&2 + exit 1 + fi + local expected actual + expected="$(expected_rocm_version 2>/dev/null || true)" + actual="$(image_rocm_version_on_target || true)" + if [ -n "$expected" ] && [ -n "$actual" ] && [ "$actual" != "$expected" ]; then + echo "::warning::Target image ROCM_VERSION=${actual} (expected ${expected}) — SDK fallback or stale image may apply" + fi + echo "::notice::Image ${DOCKER_IMAGE} present on target node (ROCM_VERSION=${actual:-unknown})." +} + +cmd_verify_rocm() { + docker_run ' + source /etc/profile.d/rocm-env.sh + echo "=== rocminfo ===" + rocminfo + echo "=== amd-smi version ===" + amd-smi version 2>/dev/null || amdsmi version 2>/dev/null || echo "amd-smi not present" + ' +} + +cmd_install_rvs() { + require_env ROCM_MAJOR + require_env INSTALL_DIR + require_env RVS_BIN + require_env REMOTE_WORK_DIR + if rvs_in_image; then + docker_run " + source /etc/profile.d/rocm-env.sh + set -euo pipefail + RVS_BIN=${RVS_BIN} + if [ ! -x \"\${RVS_BIN}\" ]; then + RVS_BIN=\$(command -v rvs || true) + fi + if [ -z \"\${RVS_BIN}\" ] || [ ! -x \"\${RVS_BIN}\" ]; then + echo \"::error::rvs binary missing in image (expected ${RVS_BIN})\" >&2 + exit 1 + fi + echo \"::notice::RVS already installed in image at \${RVS_BIN}\" + \"\${RVS_BIN}\" --version || true + " + return 0 + fi + require_env TARBALL_NAME + require_env TARBALL_URL + + if use_target_docker; then + cmd_download_rvs_tarball + else + local local_pkg="${REPO_ROOT}/pkg/${TARBALL_NAME}" + mkdir -p "${REPO_ROOT}/pkg" + if [ ! -f "${local_pkg}" ]; then + phase_start "Download RVS tarball locally" + curl -fL --max-time 600 -o "${local_pkg}" "${TARBALL_URL}" + phase_end "Download RVS tarball locally" + fi + fi + + docker_run " + source /etc/profile.d/rocm-env.sh + set -euo pipefail + PKG=/pkg/${TARBALL_NAME} + INSTALL_DIR=${INSTALL_DIR} + RVS_BIN=${RVS_BIN} + if [ ! -f \"\${PKG}\" ]; then + echo \"::error::RVS tarball not mounted at \${PKG} — run download-rvs-tarball first\" >&2 + exit 1 + fi + echo \"Installing RVS from pre-staged tarball \${PKG}...\" + mkdir -p \"\${INSTALL_DIR}\" + tar -xzf \"\${PKG}\" -C \"\${INSTALL_DIR}\" + export LD_LIBRARY_PATH=\"\${INSTALL_DIR}/lib:\${LD_LIBRARY_PATH}\" + if [ ! -x \"\${RVS_BIN}\" ]; then + echo \"::error::rvs binary missing at \${RVS_BIN}\" >&2 + exit 1 + fi + echo \"::notice::Installed RVS at \${RVS_BIN}\" + ldd \"\${RVS_BIN}\" | grep 'not found' && exit 1 || true + \"\${RVS_BIN}\" --version + " +} + +cmd_run_level4() { + require_env INSTALL_DIR + require_env RVS_BIN + require_env REMOTE_WORK_DIR + require_env ROCM_MAJOR + + if use_target_docker && ! rvs_in_image; then + require_ssh_config + if ! ssh -q -F "$SSH_CONFIG_FILE" rvs-target "test -x '${REMOTE_WORK_DIR}/extras-${ROCM_MAJOR}/bin/rvs'"; then + echo "::error::RVS not installed on target at ${REMOTE_WORK_DIR}/extras-${ROCM_MAJOR}/bin/rvs — run install-rvs first." >&2 + exit 1 + fi + elif ! rvs_in_image; then + local host_install + host_install="$(target_rvs_install_dir)" + if [ ! -x "${host_install}/bin/rvs" ]; then + echo "::error::RVS not installed at ${host_install}/bin/rvs — run install-rvs first." >&2 + exit 1 + fi + fi + + mkdir -p "${REPO_ROOT}/reports" + local start end rc + start="$(date -u +%FT%TZ)" + + set +e + docker_run " + source /etc/profile.d/rocm-env.sh + set -euo pipefail + RVS_BIN=${RVS_BIN} + if [ ! -x \"\${RVS_BIN}\" ]; then + RVS_BIN=\$(command -v rvs) + fi + export LD_LIBRARY_PATH=\"${INSTALL_DIR}/lib:\${LD_LIBRARY_PATH:-}\" + mkdir -p /reports + \"\${RVS_BIN}\" -r 4 2>&1 | tee /reports/rvs_level_4.log + exit \${PIPESTATUS[0]} + " + rc=$? + set -e + end="$(date -u +%FT%TZ)" + + if use_target_docker; then + mkdir -p "${REPO_ROOT}/reports" + scp -q -F "$SSH_CONFIG_FILE" "rvs-target:${REMOTE_WORK_DIR}/reports/rvs_level_4.log" \ + "${REPO_ROOT}/reports/" 2>/dev/null || echo "::warning::Could not copy rvs_level_4.log from target." + fi + + if [ -n "${GITHUB_OUTPUT:-}" ]; then + echo "rc=$rc" >> "$GITHUB_OUTPUT" + echo "start=$start" >> "$GITHUB_OUTPUT" + echo "end=$end" >> "$GITHUB_OUTPUT" + fi + return 0 +} + +cmd_capture_versions() { + require_env RVS_BIN + require_env INSTALL_DIR + require_env REMOTE_WORK_DIR + require_env ROCM_MAJOR + + local rvs_version target_rocm_version + rvs_version=$(docker_run " + source /etc/profile.d/rocm-env.sh + export LD_LIBRARY_PATH=\"${INSTALL_DIR}/lib:\${LD_LIBRARY_PATH}\" + ${RVS_BIN} --version 2>/dev/null | head -1 + " | tail -1) + target_rocm_version=$(docker_run " + source /etc/profile.d/rocm-env.sh + cat \"\${TARGET_ROCM_PATH}/.info/version\" 2>/dev/null \ + || cat \"\${TARGET_ROCM_PATH}/share/doc/rocm-core/version\" 2>/dev/null \ + || cat /opt/rocm/install/.info/version 2>/dev/null \ + || cat /opt/rocm/.info/version 2>/dev/null \ + || echo unknown + " | tail -1) + + if [ -n "${GITHUB_OUTPUT:-}" ]; then + echo "rvs_version=${rvs_version:-unknown}" >> "$GITHUB_OUTPUT" + echo "target_rocm_version=${target_rocm_version:-unknown}" >> "$GITHUB_OUTPUT" + fi +} + +cmd_run_pipeline() { + require_env REMOTE_WORK_DIR + cmd_verify_rocm + cmd_install_rvs + cmd_run_level4 +} + +main() { + [ $# -ge 1 ] || { usage; exit 1; } + case "$1" in + pull-image) cmd_pull_image ;; + ensure-image-on-target) cmd_ensure_image_on_target ;; + transfer-image-to-target) cmd_ensure_image_on_target ;; + build-image-on-target) cmd_build_image_on_target ;; + verify-image-on-target) cmd_verify_image_on_target ;; + download-rvs-tarball) cmd_download_rvs_tarball ;; + verify-rocm) cmd_verify_rocm ;; + install-rvs) cmd_install_rvs ;; + run-level4) cmd_run_level4 ;; + capture-versions) cmd_capture_versions ;; + run-pipeline) cmd_run_pipeline ;; + -h|--help) usage ;; + *) echo "Unknown command: $1" >&2; usage; exit 1 ;; + esac +} + +main "$@" diff --git a/rvs_nightly_test.sh b/rvs_nightly_test.sh new file mode 100755 index 000000000..6f04b94ba --- /dev/null +++ b/rvs_nightly_test.sh @@ -0,0 +1,511 @@ +#!/bin/bash +################################################################################ +# Remote RVS nightly install + test driver (used by rvs-nightly-tests.yml). +# Mirrors build_packages_local.sh: workflow orchestrates, this script executes. +################################################################################ + +set -euo pipefail + +usage() { + cat <<'EOF' +Usage: rvs_nightly_test.sh + +Commands (workflow order): + validate-config Fail fast; emit prepare job outputs to GITHUB_OUTPUT + setup-ssh Write SSH key/config under RUNNER_TEMP and verify connectivity + verify-rocm Remote rocminfo + amd-smi on TARGET_ROCM_PATH + download-tarball curl tarball to ./pkg/ on the orchestrator runner only (never on SSH target) + copy-to-target scp tarball to REMOTE_WORK_DIR/pkg on target + install-rvs Extract tarball on target under INSTALL_DIR + verify-rvs-binary Remote ldd check on RVS_BIN + run-level4 Run rvs -r 4; write rc/start/end to GITHUB_OUTPUT if set + collect-logs scp remote logs to ./reports/ + capture-versions Write rvs_version and target_rocm_version to GITHUB_OUTPUT + build-report Write ./reports/SUMMARY.md from env + prior outputs + cleanup-remote rm -rf REMOTE_WORK_DIR on target + cleanup-local-ssh Remove workflow-scoped key/config files +EOF +} + +require_env() { + local name="$1" + if [ -z "${!name:-}" ]; then + echo "::error::Required environment variable $name is not set" >&2 + exit 1 + fi +} + + +# Per-target libomp dir: ${ROCM}/lib/llvm/lib/$(clang --print-target-triple) +rvs_llvm_host_runtime_dir() { + require_env TARGET_ROCM_PATH + require_env SSH_CONFIG_FILE + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "TARGET_ROCM_PATH='${TARGET_ROCM_PATH}' bash -s" <<'REMOTE' +set -euo pipefail +clang="" +for candidate in \ + "${TARGET_ROCM_PATH}/bin/amdclang++" \ + "${TARGET_ROCM_PATH}/lib/llvm/bin/amdclang++" \ + "${TARGET_ROCM_PATH}/lib/llvm/bin/clang++"; do + [ -x "$candidate" ] && clang="$candidate" && break +done +[ -n "$clang" ] || { + echo "::warning::No ROCm clang under ${TARGET_ROCM_PATH}; skipping per-target libomp path" >&2 + exit 0 +} +triple=$("$clang" --print-target-triple 2>/dev/null) || exit 0 +dir="${TARGET_ROCM_PATH}/lib/llvm/lib/${triple}" +[ -d "$dir" ] || { + echo "::warning::llvm runtime triple dir not found under ${dir}" >&2 + exit 0 +} +echo "::notice::llvm runtime triple dir found under ${dir}" >&2 +printf '%s\n' "$dir" +REMOTE +} + +ld_path_export() { + echo "${INSTALL_DIR}/lib:${TARGET_ROCM_PATH}/lib/rocm_sysdeps/lib:${TARGET_ROCM_PATH}/lib/llvm/lib:${TARGET_ROCM_PATH}/lib" +} + +cmd_validate_config() { + require_env TARBALL_NAME + require_env TARGET_NODE + require_env TARGET_ROCM_PATH + + local input_remote="${INPUT_REMOTE_WORK_DIR:-}" + local var_remote="${VAR_REMOTE_WORK_DIR:-}" + local run_id="${GITHUB_RUN_ID:-local}" + + if [ -n "$input_remote" ]; then + REMOTE_WORK_DIR="$input_remote" + elif [ -n "$var_remote" ]; then + REMOTE_WORK_DIR="$var_remote" + else + REMOTE_WORK_DIR="/tmp/rvs-nightly-${run_id}" + fi + + if [[ "$TARBALL_NAME" =~ ^amdrocm([0-9]+)- ]]; then + ROCM_MAJOR="${BASH_REMATCH[1]}" + else + echo "::error::Cannot parse ROCm major version from tarball name: $TARBALL_NAME" >&2 + exit 1 + fi + + INSTALL_DIR="/opt/rocm/extras-${ROCM_MAJOR}" + RVS_BIN="${INSTALL_DIR}/bin/rvs" + + # Do not print orchestrator or target host identity here (hostname/IP may be + # workflow inputs and must not appear in CI logs). GitHub may still log runner + # metadata in the job "Set up job" phase. + echo "SSH target : (omitted — use secrets RVS_TARGET_* / workflow inputs)" + echo "Target ROCm path : $TARGET_ROCM_PATH" + echo "Remote work dir : $REMOTE_WORK_DIR" + echo "ROCm major : $ROCM_MAJOR" + echo "Expected RVS binary : $RVS_BIN" + + if [ -n "${GITHUB_OUTPUT:-}" ]; then + { + echo "remote_work_dir=$REMOTE_WORK_DIR" + echo "rocm_major=$ROCM_MAJOR" + echo "install_dir=$INSTALL_DIR" + echo "rvs_bin=$RVS_BIN" + } >> "$GITHUB_OUTPUT" + fi + + export REMOTE_WORK_DIR ROCM_MAJOR INSTALL_DIR RVS_BIN +} + +ssh_state_paths() { + local base="${RUNNER_TEMP:-/tmp}" + SSH_KEY_FILE="${base}/rvs_target_key" + SSH_CONFIG_FILE="${base}/rvs_target_ssh_config" + mkdir -p "$base" +} + +cmd_setup_ssh() { + require_env TARGET_NODE + require_env REMOTE_WORK_DIR + + ssh_state_paths + + if [ -n "${GITHUB_ENV:-}" ]; then + { + echo "SSH_KEY_FILE=$SSH_KEY_FILE" + echo "SSH_CONFIG_FILE=$SSH_CONFIG_FILE" + } >> "$GITHUB_ENV" + fi + + # Persist for manual runs: rvs_nightly_docker.sh sources this if SSH_CONFIG_FILE is unset. + local ssh_env_file="${RUNNER_TEMP:-/tmp}/rvs_ssh_env.sh" + cat > "$ssh_env_file" <&2 + exit 1 + fi + + install -m 600 /dev/null "$SSH_KEY_FILE" + printf '%s\n' "$SSH_PRIVATE_KEY" > "$SSH_KEY_FILE" + chmod 600 "$SSH_KEY_FILE" + + local known_hosts="${SSH_CONFIG_FILE}.known_hosts" + : > "$known_hosts" + chmod 600 "$known_hosts" + ssh-keyscan -T 15 -H "$TARGET_NODE" >> "$known_hosts" 2>/dev/null || true + + cat > "$SSH_CONFIG_FILE" <&2 + exit 1 + fi + echo "::notice::SSH connectivity OK." + + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "mkdir -p '${REMOTE_WORK_DIR}/pkg' '${REMOTE_WORK_DIR}/reports' '${REMOTE_WORK_DIR}/docker'" +} + +cmd_verify_rocm() { + require_env SSH_CONFIG_FILE + require_env TARGET_ROCM_PATH + + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "TARGET_ROCM_PATH='$TARGET_ROCM_PATH' bash -s" <<'REMOTE' +set -euo pipefail +echo "=== System ===" +# Omit nodename (uname -a prints hostname); -srvmo keeps kernel/OS/arch only. +uname -srvmo +echo +echo "=== Target ROCm path: ${TARGET_ROCM_PATH} ===" +if [ ! -d "${TARGET_ROCM_PATH}" ]; then + echo "::error::${TARGET_ROCM_PATH} does not exist on the target node." + exit 1 +fi +echo "=== ${TARGET_ROCM_PATH}/bin/rocminfo ===" +"${TARGET_ROCM_PATH}/bin/rocminfo" +echo +echo "=== ${TARGET_ROCM_PATH}/bin/amd-smi version ===" +"${TARGET_ROCM_PATH}/bin/amd-smi" version +echo +echo "::notice::ROCm prerequisites OK on target node at ${TARGET_ROCM_PATH}" +REMOTE +} + +cmd_download_tarball() { + require_env TARBALL_URL + require_env TARBALL_NAME + mkdir -p ./pkg + echo "Downloading tarball on orchestrator runner (target node is not used for this fetch):" + echo " $TARBALL_URL" + curl -fL --max-time 600 -o "./pkg/${TARBALL_NAME}" "${TARBALL_URL}" + ls -la ./pkg/ + file "./pkg/${TARBALL_NAME}" || true +} + +cmd_copy_to_target() { + require_env SSH_CONFIG_FILE + require_env TARBALL_NAME + require_env REMOTE_WORK_DIR + echo "Copying ./pkg/${TARBALL_NAME} -> rvs-target:${REMOTE_WORK_DIR}/pkg/" + scp -q -F "$SSH_CONFIG_FILE" "./pkg/${TARBALL_NAME}" \ + "rvs-target:${REMOTE_WORK_DIR}/pkg/${TARBALL_NAME}" + ssh -q -F "$SSH_CONFIG_FILE" rvs-target "ls -la '${REMOTE_WORK_DIR}/pkg/'" +} + +cmd_install_rvs() { + require_env SSH_CONFIG_FILE + require_env TARBALL_NAME + require_env REMOTE_WORK_DIR + require_env ROCM_MAJOR + require_env INSTALL_DIR + require_env RVS_BIN + require_env TARGET_ROCM_PATH + + local llvm_host_rt + llvm_host_rt=$(rvs_llvm_host_runtime_dir) + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "TARBALL_NAME='$TARBALL_NAME' REMOTE_WORK_DIR='$REMOTE_WORK_DIR' ROCM_MAJOR='$ROCM_MAJOR' INSTALL_DIR='$INSTALL_DIR' RVS_BIN='$RVS_BIN' TARGET_ROCM_PATH='$TARGET_ROCM_PATH' RVS_LLVM_HOST_RUNTIME_DIR='$llvm_host_rt' bash -s" <<'REMOTE' +set -euo pipefail +PKG="${REMOTE_WORK_DIR}/pkg/${TARBALL_NAME}" +echo "Target ROCm path : ${TARGET_ROCM_PATH}" +echo "Detected ROCm major version : ${ROCM_MAJOR}" +echo "Install target : ${INSTALL_DIR}" +echo "Expected RVS binary : ${RVS_BIN}" +if [ ! -f "$PKG" ]; then + echo "::error::Tarball not found on target node at $PKG" + exit 1 +fi +probe="$INSTALL_DIR" +while [ ! -d "$probe" ] && [ "$probe" != "/" ]; do probe=$(dirname "$probe"); done +if [ -w "$probe" ]; then + echo "Installing into $INSTALL_DIR without sudo (writable path)" + mkdir -p "$INSTALL_DIR" + tar -xzf "$PKG" -C "$INSTALL_DIR" +else + echo "Installing into $INSTALL_DIR via sudo -n (path not user-writable)" + sudo -n mkdir -p "$INSTALL_DIR" + sudo -n tar -xzf "$PKG" -C "$INSTALL_DIR" +fi +if [ ! -x "$RVS_BIN" ]; then + echo "::error::rvs binary not found or not executable at $RVS_BIN after install" + ls -la "$INSTALL_DIR/" || true + ls -la "$INSTALL_DIR/bin/" || true + exit 1 +fi +export LD_LIBRARY_PATH="${RVS_LLVM_HOST_RUNTIME_DIR:+$RVS_LLVM_HOST_RUNTIME_DIR:}${INSTALL_DIR}/lib:${TARGET_ROCM_PATH}/lib/rocm_sysdeps/lib:${TARGET_ROCM_PATH}/lib/llvm/lib:${TARGET_ROCM_PATH}/lib:${LD_LIBRARY_PATH:-}" +echo "Installed RVS at: $RVS_BIN" +"$RVS_BIN" --version || true +REMOTE +} + +cmd_verify_rvs_binary() { + require_env SSH_CONFIG_FILE + require_env RVS_BIN + require_env INSTALL_DIR + require_env TARGET_ROCM_PATH + + local llvm_host_rt + llvm_host_rt=$(rvs_llvm_host_runtime_dir) + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "RVS_BIN='$RVS_BIN' INSTALL_DIR='$INSTALL_DIR' TARGET_ROCM_PATH='$TARGET_ROCM_PATH' RVS_LLVM_HOST_RUNTIME_DIR='$llvm_host_rt' bash -s" <<'REMOTE' +set -euo pipefail +if [ ! -x "$RVS_BIN" ]; then + echo "::error::$RVS_BIN was not produced by tarball extraction on target." + exit 1 +fi +export LD_LIBRARY_PATH="${RVS_LLVM_HOST_RUNTIME_DIR:+$RVS_LLVM_HOST_RUNTIME_DIR:}${INSTALL_DIR}/lib:${TARGET_ROCM_PATH}/lib/rocm_sysdeps/lib:${TARGET_ROCM_PATH}/lib/llvm/lib:${TARGET_ROCM_PATH}/lib:${LD_LIBRARY_PATH:-}" +echo "=== ldd $RVS_BIN ===" +LDD_OUTPUT=$(ldd "$RVS_BIN" 2>&1 || true) +echo "$LDD_OUTPUT" +if echo "$LDD_OUTPUT" | grep -q "not found"; then + echo "::error::RVS binary has unresolved library dependencies on target (see above)." + exit 1 +fi +echo "::notice::RVS binary's library dependencies resolved OK on target." +REMOTE +} + +cmd_run_level4() { + require_env SSH_CONFIG_FILE + require_env RVS_BIN + require_env INSTALL_DIR + require_env REMOTE_WORK_DIR + require_env TARGET_ROCM_PATH + + set +e + local start end rc + start=$(date -u +%FT%TZ) + echo "::group::RVS level 4 (${RVS_BIN} -r 4) against ${TARGET_ROCM_PATH}" + local llvm_host_rt + llvm_host_rt=$(rvs_llvm_host_runtime_dir) + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "RVS_BIN='$RVS_BIN' INSTALL_DIR='$INSTALL_DIR' REMOTE_WORK_DIR='$REMOTE_WORK_DIR' TARGET_ROCM_PATH='$TARGET_ROCM_PATH' RVS_LLVM_HOST_RUNTIME_DIR='$llvm_host_rt' bash -s" <<'REMOTE' +set +e +export LD_LIBRARY_PATH="${RVS_LLVM_HOST_RUNTIME_DIR:+$RVS_LLVM_HOST_RUNTIME_DIR:}${INSTALL_DIR}/lib:${TARGET_ROCM_PATH}/lib/rocm_sysdeps/lib:${TARGET_ROCM_PATH}/lib/llvm/lib:${TARGET_ROCM_PATH}/lib:${LD_LIBRARY_PATH:-}" +mkdir -p "${REMOTE_WORK_DIR}/reports" +"$RVS_BIN" -r 4 2>&1 | tee "${REMOTE_WORK_DIR}/reports/rvs_level_4.log" +RC=${PIPESTATUS[0]} +echo "remote_rc=$RC" +exit $RC +REMOTE + rc=$? + echo "::endgroup::" + end=$(date -u +%FT%TZ) + + if [ -n "${GITHUB_OUTPUT:-}" ]; then + echo "rc=$rc" >> "$GITHUB_OUTPUT" + echo "start=$start" >> "$GITHUB_OUTPUT" + echo "end=$end" >> "$GITHUB_OUTPUT" + fi + # Caller workflow decides whether to fail the job on non-zero rc. + return 0 +} + +cmd_collect_logs() { + require_env SSH_CONFIG_FILE + require_env REMOTE_WORK_DIR + mkdir -p ./reports + if [ ! -f "$SSH_CONFIG_FILE" ]; then + echo "::warning::SSH config not present; skipping log collection." + return 0 + fi + echo "Copying logs from rvs-target:${REMOTE_WORK_DIR}/reports/ -> ./reports/" + scp -q -F "$SSH_CONFIG_FILE" "rvs-target:${REMOTE_WORK_DIR}/reports/*.log" ./reports/ || \ + echo "::warning::No log files retrieved from target node." + ls -la ./reports/ || true +} + +cmd_capture_versions() { + require_env SSH_CONFIG_FILE + require_env RVS_BIN + require_env INSTALL_DIR + require_env TARGET_ROCM_PATH + + local llvm_host_rt + llvm_host_rt=$(rvs_llvm_host_runtime_dir) + local rvs_version target_rocm_version + rvs_version=$( + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "RVS_BIN='$RVS_BIN' INSTALL_DIR='$INSTALL_DIR' TARGET_ROCM_PATH='$TARGET_ROCM_PATH' RVS_LLVM_HOST_RUNTIME_DIR='$llvm_host_rt' bash -s" <<'REMOTE' 2>/dev/null | head -1 || echo "unknown" +export LD_LIBRARY_PATH="${RVS_LLVM_HOST_RUNTIME_DIR:+$RVS_LLVM_HOST_RUNTIME_DIR:}${INSTALL_DIR}/lib:${TARGET_ROCM_PATH}/lib/rocm_sysdeps/lib:${TARGET_ROCM_PATH}/lib/llvm/lib:${TARGET_ROCM_PATH}/lib:${LD_LIBRARY_PATH:-}" +"$RVS_BIN" --version 2>/dev/null +REMOTE + ) + target_rocm_version=$( + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "TARGET_ROCM_PATH='$TARGET_ROCM_PATH' bash -s" <<'REMOTE' 2>/dev/null || echo "unknown" +cat "${TARGET_ROCM_PATH}/.info/version" 2>/dev/null \ + || cat "${TARGET_ROCM_PATH}/share/doc/rocm-core/version" 2>/dev/null \ + || echo "unknown" +REMOTE + ) + + if [ -n "${GITHUB_OUTPUT:-}" ]; then + echo "rvs_version=$rvs_version" >> "$GITHUB_OUTPUT" + echo "target_rocm_version=$target_rocm_version" >> "$GITHUB_OUTPUT" + fi +} + +cmd_build_report() { + require_env TARBALL_NAME + require_env TARGET_ROCM_PATH + require_env REMOTE_WORK_DIR + require_env RVS_BIN + + # TARBALL_URL is optional: the report job reads it from install-rvs-on-target outputs. + # If that output was empty (e.g. legacy echo-to-GITHUB_OUTPUT with special URL chars), + # still emit SUMMARY.md with the tarball filename. + local tarball_url_display="${TARBALL_URL:-}" + if [ -z "$tarball_url_display" ]; then + tarball_url_display="_(unavailable — check install job resolve step / use heredoc GITHUB_OUTPUT for URLs with special characters)_" + fi + + local rc4="${RVS_LEVEL4_RC:-0}" + local start="${RVS_LEVEL4_START:-}" + local end="${RVS_LEVEL4_END:-}" + local rvs_version="${RVS_VERSION:-unknown}" + local target_rocm_version="${TARGET_ROCM_VERSION:-unknown}" + local run_id="${GITHUB_RUN_ID:-local}" + local server_url="${GITHUB_SERVER_URL:-https://github.com}" + local repository="${GITHUB_REPOSITORY:-local/repo}" + local event_name="${GITHUB_EVENT_NAME:-local}" + + mkdir -p ./reports + local s4 overall + if [ "$rc4" -eq 0 ]; then + s4=PASS + overall=PASS + else + s4=FAIL + overall=FAIL + fi + + local report=./reports/SUMMARY.md + { + echo "# RVS Nightly Test Report" + echo "" + echo "| Field | Value |" + echo "|---|---|" + echo "| Run | [\`${run_id}\`](${server_url}/${repository}/actions/runs/${run_id}) |" + echo "| Trigger | \`${event_name}\` |" + echo "| Target ROCm path | \`${TARGET_ROCM_PATH}\` (version \`${target_rocm_version}\`) |" + echo "| Remote work dir | \`${REMOTE_WORK_DIR}\` |" + echo "| Tarball | \`${TARBALL_NAME}\` |" + echo "| Source URL | ${tarball_url_display} |" + echo "| RVS version | \`${rvs_version}\` |" + echo "| Overall result | **${overall}** |" + echo "" + echo "## Results" + echo "" + echo "| Test | Command | Result | Exit | Started (UTC) | Ended (UTC) |" + echo "|---|---|:--:|---:|---|---|" + echo "| Level 4 | \`${RVS_BIN} -r 4\` | ${s4} | ${rc4} | ${start} | ${end} |" + echo "" + echo "## Logs" + echo "" + echo "Full stdout/stderr from the level-4 run is attached to the artifact" + echo "\`rvs-nightly-report-${run_id}\`:" + echo "" + echo "- \`rvs_level_4.log\`" + echo "- \`SUMMARY.md\` (this file)" + } > "$report" + + cat "$report" + if [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + cat "$report" >> "$GITHUB_STEP_SUMMARY" + fi + + if [ -n "${GITHUB_OUTPUT:-}" ]; then + echo "overall=$overall" >> "$GITHUB_OUTPUT" + echo "rc4=$rc4" >> "$GITHUB_OUTPUT" + fi +} + +cmd_cleanup_remote() { + set +e + if [ -z "${REMOTE_WORK_DIR:-}" ] || [ -z "${SSH_CONFIG_FILE:-}" ] || [ ! -f "$SSH_CONFIG_FILE" ]; then + return 0 + fi + ssh -q -F "$SSH_CONFIG_FILE" rvs-target "rm -rf '${REMOTE_WORK_DIR}'" || \ + echo "::warning::Failed to clean up ${REMOTE_WORK_DIR} on target node." +} + +cmd_cleanup_local_ssh() { + set +e + if [ -n "${SSH_KEY_FILE:-}" ]; then + rm -f "$SSH_KEY_FILE" + fi + if [ -n "${SSH_CONFIG_FILE:-}" ]; then + rm -f "$SSH_CONFIG_FILE" "${SSH_CONFIG_FILE}.known_hosts" + fi +} + +main() { + if [ $# -lt 1 ]; then + usage + exit 1 + fi + case "$1" in + validate-config) cmd_validate_config ;; + setup-ssh) cmd_setup_ssh ;; + verify-rocm) cmd_verify_rocm ;; + download-tarball) cmd_download_tarball ;; + copy-to-target) cmd_copy_to_target ;; + install-rvs) cmd_install_rvs ;; + verify-rvs-binary) cmd_verify_rvs_binary ;; + run-level4) cmd_run_level4 ;; + collect-logs) cmd_collect_logs ;; + capture-versions) cmd_capture_versions ;; + build-report) cmd_build_report ;; + cleanup-remote) cmd_cleanup_remote ;; + cleanup-local-ssh) cmd_cleanup_local_ssh ;; + -h|--help) usage ;; + *) + echo "::error::Unknown command: $1" >&2 + usage + exit 1 + ;; + esac +} + +main "$@" diff --git a/rvs_pr_test.sh b/rvs_pr_test.sh new file mode 100755 index 000000000..42a48be6f --- /dev/null +++ b/rvs_pr_test.sh @@ -0,0 +1,782 @@ +#!/bin/bash +################################################################################ +# Remote RVS PR-package install + test driver (used by rvs-pr-tests.yml). +# Like rvs_nightly_test.sh but defaults to /tmp/rvs-pr-* staging; primary artifact +# is the amdrocm*-rvs-*-Linux.tar.gz relocatable tarball (optional .deb still supported). +################################################################################ + +set -euo pipefail + +usage() { + cat <<'EOF' +Usage: rvs_pr_test.sh + +Commands (workflow order): + resolve-package-url Discover latest Linux .tar.gz from rvs/ root, from a .../manylinux_*/ listing dir, or a direct .tar.gz / .deb URL; + writes tarball_url + tarball_name to GITHUB_OUTPUT / GITHUB_ENV when set + peek-latest-artifact-key Print one-line artifact key (build/pr/tgz, tgz:basename, or deb:basename) for cache compare; stdout only + validate-config Fail fast; emit prepare job outputs to GITHUB_OUTPUT + setup-ssh Write SSH key/config under RUNNER_TEMP and verify connectivity + verify-rocm Remote rocminfo + amd-smi on TARGET_ROCM_PATH + download-tarball curl package (.deb or .tar.gz) to ./pkg/ on the orchestrator only + copy-to-target scp package to REMOTE_WORK_DIR/pkg on target + install-rvs Install .deb via dpkg on target, or extract .tar.gz under INSTALL_DIR + verify-rvs-binary Remote ldd check on RVS_BIN + run-level4 Run rvs -r 4; write rc/start/end to GITHUB_OUTPUT if set + collect-logs scp remote logs to ./reports/ + capture-versions Write rvs_version and target_rocm_version to GITHUB_OUTPUT + build-report Write ./reports/SUMMARY.md from env + prior outputs + cleanup-remote rm -rf REMOTE_WORK_DIR on target + cleanup-local-ssh Remove workflow-scoped key/config files +EOF +} + +require_env() { + local name="$1" + if [ -z "${!name:-}" ]; then + echo "::error::Required environment variable $name is not set" >&2 + exit 1 + fi +} + + +# Per-target libomp dir on the SSH target: ${ROCM}/lib/llvm/lib/$(clang --print-target-triple) +# TARGET_ROCM_PATH is a path on the target node, not on this orchestrator runner. +rvs_llvm_host_runtime_dir() { + require_env TARGET_ROCM_PATH + require_env SSH_CONFIG_FILE + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "TARGET_ROCM_PATH='${TARGET_ROCM_PATH}' bash -s" <<'REMOTE' +set -euo pipefail +clang="" +for candidate in \ + "${TARGET_ROCM_PATH}/bin/amdclang++" \ + "${TARGET_ROCM_PATH}/lib/llvm/bin/amdclang++" \ + "${TARGET_ROCM_PATH}/lib/llvm/bin/clang++"; do + [ -x "$candidate" ] && clang="$candidate" && break +done +[ -n "$clang" ] || { + echo "::warning::No ROCm clang under ${TARGET_ROCM_PATH}; skipping per-target libomp path" >&2 + exit 0 +} +triple=$("$clang" --print-target-triple 2>/dev/null) || exit 0 +dir="${TARGET_ROCM_PATH}/lib/llvm/lib/${triple}" +[ -d "$dir" ] || { + echo "::warning::llvm runtime triple dir not found under ${dir}" >&2 + exit 0 +} +echo "::notice::llvm runtime triple dir found under ${dir}" >&2 +printf '%s\n' "$dir" +REMOTE +} + + +ld_path_export() { + echo "${INSTALL_DIR}/lib:${TARGET_ROCM_PATH}/lib/rocm_sysdeps/lib:${TARGET_ROCM_PATH}/lib/llvm/lib:${TARGET_ROCM_PATH}/lib" +} + +# --- CDN rvs/ root → latest Linux tarball (orchestrator only; curl never on GPU target) --- + +cdn_curl_listing() { + local url="$1" + curl -sSL --max-time 120 --retry 2 --retry-delay 2 \ + -A 'Mozilla/5.0 (compatible; RVS-PR-Tests/1.0)' "$url" 2>/dev/null || true +} + +# Largest purely-numeric directory name from an S3/CF HTML index (href="1234/"). +cdn_latest_numeric_dir() { + local html="$1" + printf '%s' "$html" | grep -oE 'href="[0-9]+/"' | sed 's/href="//;s/\/"//' | sort -n | tail -n 1 || true +} + +# First subdirectory (non-parent) containing an amdrocm*-rvs-*-Linux.tar.gz link; prefer manylinux_*. +cdn_find_tgz_listing_url() { + local pr_page_html="$1" + local pr_base="$2" + local line dir html pick + pick="" + while IFS= read -r line; do + dir="${line%/}" + [[ -z "$dir" || "$dir" == '..' ]] && continue + html="$(cdn_curl_listing "${pr_base}${dir}/")" + if ! printf '%s' "$html" | grep -qE 'amdrocm[0-9]+-rvs-[0-9A-Za-z._\-]+-Linux\.tar\.gz'; then + continue + fi + if [[ "$dir" == manylinux* ]]; then + pick="$dir" + break + fi + if [ -z "$pick" ]; then + pick="$dir" + fi + done < <(printf '%s' "$pr_page_html" | grep -oE 'href="[^./][^"]*/"' | sed 's/href="//;s/"$//' | grep -v '^\.\./$' || true) + + if [ -z "$pick" ]; then + echo "" + return 0 + fi + printf '%s' "${pr_base}${pick}/" +} + +# Args: internal rvs root URL without trailing slash (not a direct .tar.gz / .deb file). +# Prints: buildprtgz_nametgz_https_url to stdout; errors to stderr; exit 1 on failure. +cdn_resolve_latest_merge_tgz_metadata() { + local root="$1" + local base="${root}/" + local idx html build merge_base merge_html pr pr_page tgz_url tgz_dir_html tgz_name + + idx="$(cdn_curl_listing "$base")" + if [ -z "$idx" ]; then + echo "::error::Empty listing from ${base} (check URL and orchestrator HTTPS egress)." >&2 + return 1 + fi + build="$(cdn_latest_numeric_dir "$idx")" + if [ -z "$build" ]; then + echo "::error::No numeric build directory under ${base}" >&2 + return 1 + fi + html="$(cdn_curl_listing "${base}${build}/")" + if ! printf '%s' "$html" | grep -q 'href="merge/"'; then + echo "::error::Build ${build} has no merge/ directory (expected rvs//merge/...)." >&2 + return 1 + fi + merge_base="${base}${build}/merge/" + merge_html="$(cdn_curl_listing "$merge_base")" + pr="$(cdn_latest_numeric_dir "$merge_html")" + if [ -z "$pr" ]; then + echo "::error::No numeric PR directory under ${merge_base}" >&2 + return 1 + fi + pr_page="$(cdn_curl_listing "${merge_base}${pr}/")" + tgz_url="$(cdn_find_tgz_listing_url "$pr_page" "${merge_base}${pr}/")" + if [ -z "$tgz_url" ]; then + echo "::error::No subdirectory under ${merge_base}${pr}/ listing amdrocm*-rvs-*-Linux.tar.gz." >&2 + return 1 + fi + tgz_dir_html="$(cdn_curl_listing "$tgz_url")" + tgz_name="$(printf '%s' "$tgz_dir_html" | grep -oE 'amdrocm[0-9]+-rvs-[0-9A-Za-z._\-]+-Linux\.tar\.gz' 2>/dev/null | sort -uV | tail -n 1 || true)" + if [ -z "$tgz_name" ]; then + echo "::error::Could not parse tarball filename under ${tgz_url}" >&2 + return 1 + fi + printf '%s\t%s\t%s\t%s' "$build" "$pr" "$tgz_name" "${tgz_url}${tgz_name}" +} + +# HTTPS directory listing ending in .../manylinux_*/ (e.g. PR build upload path). +# Prints: tarball_https_urlbasename to stdout; errors to stderr; exit 1 on failure. +cdn_resolve_tgz_from_manylinux_dir() { + local root="$1" + root="${root%/}" + local base="${root}/" + local html tgz_name + + html="$(cdn_curl_listing "$base")" + if [ -z "$html" ]; then + echo "::error::Empty listing from ${base} (check URL and orchestrator HTTPS egress)." >&2 + return 1 + fi + tgz_name="$(printf '%s' "$html" | grep -oE 'amdrocm[0-9]+-rvs-[0-9A-Za-z._\-]+-Linux\.tar\.gz' 2>/dev/null | sort -uV | tail -n 1 || true)" + if [ -z "$tgz_name" ]; then + echo "::error::No amdrocm*-rvs-*-Linux.tar.gz link under ${base}" >&2 + return 1 + fi + printf '%s\t%s' "${base}${tgz_name}" "$tgz_name" +} + +cmd_peek_latest_artifact_key() { + local root="${PR_PACKAGE_ROOT_URL:-}" + if [ -z "$root" ]; then + root="${1:-}" + fi + if [ -z "${root:-}" ]; then + echo "::error::Set PR_PACKAGE_ROOT_URL (HTTPS …/rvs index root or direct .tar.gz / .deb URL)." >&2 + exit 1 + fi + root="${root%/}" + + if [[ "$root" == *.deb ]]; then + printf 'deb:%s\n' "$(basename "${root%%\?*}")" + return 0 + fi + + if [[ "$root" == *-Linux.tar.gz ]] || [[ "$root" == *.tar.gz ]]; then + printf 'tgz:%s\n' "$(basename "${root%%\?*}")" + return 0 + fi + + local meta + if ! meta="$(cdn_resolve_latest_merge_tgz_metadata "$root")"; then + exit 1 + fi + local build pr tgz_name _url + IFS=$'\t' read -r build pr tgz_name _url <<<"$meta" + printf '%s/%s/%s\n' "$build" "$pr" "$tgz_name" +} + +cmd_resolve_package_url() { + local root="${PR_PACKAGE_ROOT_URL:-}" + if [ -z "$root" ]; then + root="${1:-}" + fi + if [ -z "${root:-}" ]; then + echo "::error::Set PR_PACKAGE_ROOT_URL (HTTPS …/rvs index root or direct .tar.gz / .deb URL)." >&2 + exit 1 + fi + + root="${root%/}" + local URL NAME + + # Direct HTTPS URL to relocatable Linux tarball (same artifact as nightly). + if [[ "$root" == *-Linux.tar.gz ]] || [[ "$root" == *.tar.gz ]]; then + URL="$root" + NAME="$(basename "${URL%%\?*}")" + echo "::notice::Using direct Linux tarball URL (basename: ${NAME})" + if ! curl -fsSIL --max-time 120 -A 'Mozilla/5.0 (compatible; RVS-PR-Tests/1.0)' -o /dev/null "$URL"; then + echo "::error::Tarball URL is not retrievable: ${URL}" >&2 + exit 1 + fi + elif [[ "$root" == *.deb ]]; then + URL="$root" + NAME="$(basename "${URL%%\?*}")" + echo "::notice::Using direct .deb URL (basename: ${NAME})" + if ! curl -fsSIL --max-time 120 -A 'Mozilla/5.0 (compatible; RVS-PR-Tests/1.0)' -o /dev/null "$URL"; then + echo "::error::Deb URL is not retrievable: ${URL}" >&2 + exit 1 + fi + elif printf '%s' "$root" | grep -qE '/manylinux[^/]*$'; then + local meta_ml + echo "::notice::Resolving Linux tarball under manylinux listing: ${root}/" + if ! meta_ml="$(cdn_resolve_tgz_from_manylinux_dir "$root")"; then + exit 1 + fi + IFS=$'\t' read -r URL NAME <<<"$meta_ml" + echo "::notice::Resolved tarball: ${NAME}" + if ! curl -fsSIL --max-time 120 -A 'Mozilla/5.0 (compatible; RVS-PR-Tests/1.0)' -o /dev/null "$URL"; then + echo "::error::Resolved tarball URL is not retrievable: ${URL}" >&2 + exit 1 + fi + else + local base="${root}/" + echo "::notice::Resolving latest Linux tarball under CDN root: ${base}" + local meta build pr tgz_name tgz_url + if ! meta="$(cdn_resolve_latest_merge_tgz_metadata "$root")"; then + exit 1 + fi + IFS=$'\t' read -r build pr tgz_name tgz_url <<<"$meta" + echo "::notice::Latest build id: ${build}" + echo "::notice::Latest merge PR directory: ${pr}" + URL="$tgz_url" + NAME="$tgz_name" + echo "::notice::Resolved tarball: ${NAME}" + if ! curl -fsSIL --max-time 120 -A 'Mozilla/5.0 (compatible; RVS-PR-Tests/1.0)' -o /dev/null "$URL"; then + echo "::error::Resolved tarball URL is not retrievable: ${URL}" >&2 + exit 1 + fi + fi + + if [[ "$NAME" != *.deb ]] && [[ "$NAME" != *-Linux.tar.gz ]]; then + echo "::error::RVS PR Tests expected a *-Linux.tar.gz relocatable tarball or a .deb; got: ${NAME}" >&2 + exit 1 + fi + + if [ -n "${GITHUB_OUTPUT:-}" ]; then + # Heredoc for tarball_url (URLs may contain %, &, etc.); plain assignment for tarball_name + # (safe filename) — same pattern as rvs-nightly-tests.yml / rvs_nightly_test.sh. + { + echo "tarball_url<> "$GITHUB_OUTPUT" + fi + + if [ -n "${GITHUB_ENV:-}" ]; then + { + echo "TARBALL_URL<> "$GITHUB_ENV" + fi + + # Artifact fallback for create-test-report when job outputs drop URL/name between jobs. + mkdir -p ./pkg-meta + printf '%s\n' "$NAME" > ./pkg-meta/tarball_name + printf '%s\n' "$URL" > ./pkg-meta/tarball_url + + echo "Package file : $NAME" + echo "URL : $URL" +} + +cmd_validate_config() { + require_env TARBALL_NAME + require_env TARGET_NODE + require_env TARGET_ROCM_PATH + + local input_remote="${INPUT_REMOTE_WORK_DIR:-}" + local var_remote="${VAR_REMOTE_WORK_DIR:-}" + local run_id="${GITHUB_RUN_ID:-local}" + + if [ -n "$input_remote" ]; then + REMOTE_WORK_DIR="$input_remote" + elif [ -n "$var_remote" ]; then + REMOTE_WORK_DIR="$var_remote" + else + REMOTE_WORK_DIR="/tmp/rvs-pr-${run_id}" + fi + + # amdrocm7-rvs-…-Linux.tar.gz or amdrocm7-rvs_….deb + if [[ "$TARBALL_NAME" =~ ^amdrocm([0-9]+)- ]]; then + ROCM_MAJOR="${BASH_REMATCH[1]}" + else + echo "::error::Cannot parse ROCm major version from package name: $TARBALL_NAME" >&2 + exit 1 + fi + + INSTALL_DIR="/opt/rocm/extras-${ROCM_MAJOR}" + RVS_BIN="${INSTALL_DIR}/bin/rvs" + + # Do not print orchestrator or target host identity here (hostname/IP may be + # workflow inputs and must not appear in CI logs). GitHub may still log runner + # metadata in the job "Set up job" phase. + echo "SSH target : (omitted — use secrets RVS_TARGET_* / workflow inputs)" + echo "Target ROCm path : $TARGET_ROCM_PATH" + echo "Remote work dir : $REMOTE_WORK_DIR" + echo "ROCm major : $ROCM_MAJOR" + echo "Expected RVS binary : $RVS_BIN" + + if [ -n "${GITHUB_OUTPUT:-}" ]; then + { + echo "remote_work_dir=$REMOTE_WORK_DIR" + echo "rocm_major=$ROCM_MAJOR" + echo "install_dir=$INSTALL_DIR" + echo "rvs_bin=$RVS_BIN" + } >> "$GITHUB_OUTPUT" + fi + + export REMOTE_WORK_DIR ROCM_MAJOR INSTALL_DIR RVS_BIN +} + +ssh_state_paths() { + local base="${RUNNER_TEMP:-/tmp}" + SSH_KEY_FILE="${base}/rvs_target_key" + SSH_CONFIG_FILE="${base}/rvs_target_ssh_config" + mkdir -p "$base" +} + +cmd_setup_ssh() { + require_env TARGET_NODE + require_env REMOTE_WORK_DIR + + ssh_state_paths + + if [ -n "${GITHUB_ENV:-}" ]; then + { + echo "SSH_KEY_FILE=$SSH_KEY_FILE" + echo "SSH_CONFIG_FILE=$SSH_CONFIG_FILE" + } >> "$GITHUB_ENV" + fi + + if [ -z "${SSH_PRIVATE_KEY:-}" ]; then + echo "::error::SSH_PRIVATE_KEY is not set; cannot SSH to target node." >&2 + exit 1 + fi + + install -m 600 /dev/null "$SSH_KEY_FILE" + printf '%s\n' "$SSH_PRIVATE_KEY" > "$SSH_KEY_FILE" + chmod 600 "$SSH_KEY_FILE" + + local known_hosts="${SSH_CONFIG_FILE}.known_hosts" + : > "$known_hosts" + chmod 600 "$known_hosts" + ssh-keyscan -T 15 -H "$TARGET_NODE" >> "$known_hosts" 2>/dev/null || true + + cat > "$SSH_CONFIG_FILE" <&2 + exit 1 + fi + echo "::notice::SSH connectivity OK." + + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "mkdir -p '${REMOTE_WORK_DIR}/pkg' '${REMOTE_WORK_DIR}/reports'" +} + +cmd_verify_rocm() { + require_env SSH_CONFIG_FILE + require_env TARGET_ROCM_PATH + + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "TARGET_ROCM_PATH='$TARGET_ROCM_PATH' bash -s" <<'REMOTE' +set -euo pipefail +echo "=== System ===" +# Omit nodename (uname -a prints hostname); -srvmo keeps kernel/OS/arch only. +uname -srvmo +echo +echo "=== Target ROCm path: ${TARGET_ROCM_PATH} ===" +if [ ! -d "${TARGET_ROCM_PATH}" ]; then + echo "::error::${TARGET_ROCM_PATH} does not exist on the target node." + exit 1 +fi +echo "=== ${TARGET_ROCM_PATH}/bin/rocminfo ===" +"${TARGET_ROCM_PATH}/bin/rocminfo" +echo +echo "=== ${TARGET_ROCM_PATH}/bin/amd-smi version ===" +"${TARGET_ROCM_PATH}/bin/amd-smi" version +echo +echo "::notice::ROCm prerequisites OK on target node at ${TARGET_ROCM_PATH}" +REMOTE +} + +cmd_download_tarball() { + require_env TARBALL_URL + require_env TARBALL_NAME + mkdir -p ./pkg + echo "Downloading RVS package on orchestrator runner (target node is not used for this fetch):" + echo " $TARBALL_URL" + curl -fL --max-time 600 -o "./pkg/${TARBALL_NAME}" "${TARBALL_URL}" + ls -la ./pkg/ + file "./pkg/${TARBALL_NAME}" || true +} + +cmd_copy_to_target() { + require_env SSH_CONFIG_FILE + require_env TARBALL_NAME + require_env REMOTE_WORK_DIR + echo "Copying ./pkg/${TARBALL_NAME} -> rvs-target:${REMOTE_WORK_DIR}/pkg/" + scp -q -F "$SSH_CONFIG_FILE" "./pkg/${TARBALL_NAME}" \ + "rvs-target:${REMOTE_WORK_DIR}/pkg/${TARBALL_NAME}" + ssh -q -F "$SSH_CONFIG_FILE" rvs-target "ls -la '${REMOTE_WORK_DIR}/pkg/'" +} + +cmd_install_rvs() { + require_env SSH_CONFIG_FILE + require_env TARBALL_NAME + require_env REMOTE_WORK_DIR + require_env ROCM_MAJOR + require_env INSTALL_DIR + require_env RVS_BIN + require_env TARGET_ROCM_PATH + + local llvm_host_rt + llvm_host_rt=$(rvs_llvm_host_runtime_dir) + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "TARBALL_NAME='$TARBALL_NAME' REMOTE_WORK_DIR='$REMOTE_WORK_DIR' ROCM_MAJOR='$ROCM_MAJOR' INSTALL_DIR='$INSTALL_DIR' RVS_BIN='$RVS_BIN' TARGET_ROCM_PATH='$TARGET_ROCM_PATH' RVS_LLVM_HOST_RUNTIME_DIR='$llvm_host_rt' bash -s" <<'REMOTE' +set -euo pipefail +PKG="${REMOTE_WORK_DIR}/pkg/${TARBALL_NAME}" +echo "Target ROCm path : ${TARGET_ROCM_PATH}" +echo "Detected ROCm major version : ${ROCM_MAJOR}" +echo "Install target : ${INSTALL_DIR}" +echo "Expected RVS binary : ${RVS_BIN}" +if [ ! -f "$PKG" ]; then + echo "::error::Package file not found on target node at $PKG" + exit 1 +fi +if [[ "$TARBALL_NAME" == *.deb ]]; then + echo "Installing .deb with dpkg (non-interactive)..." + export DEBIAN_FRONTEND=noninteractive + if sudo -n dpkg -i "$PKG"; then + : + else + echo "::warning::dpkg -i had a non-zero exit; trying apt-get -f install then dpkg again" + sudo -n apt-get update -qq || true + sudo -n apt-get install -f -y -qq || true + sudo -n dpkg -i "$PKG" + fi +else + probe="$INSTALL_DIR" + while [ ! -d "$probe" ] && [ "$probe" != "/" ]; do probe=$(dirname "$probe"); done + if [ -w "$probe" ]; then + echo "Installing into $INSTALL_DIR without sudo (writable path)" + mkdir -p "$INSTALL_DIR" + tar -xzf "$PKG" -C "$INSTALL_DIR" + else + echo "Installing into $INSTALL_DIR via sudo -n (path not user-writable)" + sudo -n mkdir -p "$INSTALL_DIR" + sudo -n tar -xzf "$PKG" -C "$INSTALL_DIR" + fi +fi +if [ ! -x "$RVS_BIN" ]; then + echo "::error::rvs binary not found or not executable at $RVS_BIN after install" + ls -la "$INSTALL_DIR/" || true + ls -la "$INSTALL_DIR/bin/" || true + exit 1 +fi +export LD_LIBRARY_PATH="${RVS_LLVM_HOST_RUNTIME_DIR:+$RVS_LLVM_HOST_RUNTIME_DIR:}${INSTALL_DIR}/lib:${TARGET_ROCM_PATH}/lib/rocm_sysdeps/lib:${TARGET_ROCM_PATH}/lib/llvm/lib:${TARGET_ROCM_PATH}/lib:${LD_LIBRARY_PATH:-}" +echo "Installed RVS at: $RVS_BIN" +"$RVS_BIN" --version || true +REMOTE +} + +cmd_verify_rvs_binary() { + require_env SSH_CONFIG_FILE + require_env RVS_BIN + require_env INSTALL_DIR + require_env TARGET_ROCM_PATH + + local llvm_host_rt + llvm_host_rt=$(rvs_llvm_host_runtime_dir) + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "RVS_BIN='$RVS_BIN' INSTALL_DIR='$INSTALL_DIR' TARGET_ROCM_PATH='$TARGET_ROCM_PATH' RVS_LLVM_HOST_RUNTIME_DIR='$llvm_host_rt' bash -s" <<'REMOTE' +set -euo pipefail +if [ ! -x "$RVS_BIN" ]; then + echo "::error::$RVS_BIN was not produced by install on target." + exit 1 +fi +export LD_LIBRARY_PATH="${RVS_LLVM_HOST_RUNTIME_DIR:+$RVS_LLVM_HOST_RUNTIME_DIR:}${INSTALL_DIR}/lib:${TARGET_ROCM_PATH}/lib/rocm_sysdeps/lib:${TARGET_ROCM_PATH}/lib/llvm/lib:${TARGET_ROCM_PATH}/lib:${LD_LIBRARY_PATH:-}" +echo "=== ldd $RVS_BIN ===" +LDD_OUTPUT=$(ldd "$RVS_BIN" 2>&1 || true) +echo "$LDD_OUTPUT" +if echo "$LDD_OUTPUT" | grep -q "not found"; then + echo "::error::RVS binary has unresolved library dependencies on target (see above)." + exit 1 +fi +echo "::notice::RVS binary's library dependencies resolved OK on target." +REMOTE +} + +cmd_run_level4() { + require_env SSH_CONFIG_FILE + require_env RVS_BIN + require_env INSTALL_DIR + require_env REMOTE_WORK_DIR + require_env TARGET_ROCM_PATH + + set +e + local start end rc + start=$(date -u +%FT%TZ) + echo "::group::RVS level 4 (${RVS_BIN} -r 4) against ${TARGET_ROCM_PATH}" + local llvm_host_rt + llvm_host_rt=$(rvs_llvm_host_runtime_dir) + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "RVS_BIN='$RVS_BIN' INSTALL_DIR='$INSTALL_DIR' REMOTE_WORK_DIR='$REMOTE_WORK_DIR' TARGET_ROCM_PATH='$TARGET_ROCM_PATH' RVS_LLVM_HOST_RUNTIME_DIR='$llvm_host_rt' bash -s" <<'REMOTE' +set +e +export LD_LIBRARY_PATH="${RVS_LLVM_HOST_RUNTIME_DIR:+$RVS_LLVM_HOST_RUNTIME_DIR:}${INSTALL_DIR}/lib:${TARGET_ROCM_PATH}/lib/rocm_sysdeps/lib:${TARGET_ROCM_PATH}/lib/llvm/lib:${TARGET_ROCM_PATH}/lib:${LD_LIBRARY_PATH:-}" +mkdir -p "${REMOTE_WORK_DIR}/reports" +"$RVS_BIN" -r 4 2>&1 | tee "${REMOTE_WORK_DIR}/reports/rvs_level_4.log" +RC=${PIPESTATUS[0]} +echo "remote_rc=$RC" +exit $RC +REMOTE + rc=$? + echo "::endgroup::" + end=$(date -u +%FT%TZ) + + if [ -n "${GITHUB_OUTPUT:-}" ]; then + echo "rc=$rc" >> "$GITHUB_OUTPUT" + echo "start=$start" >> "$GITHUB_OUTPUT" + echo "end=$end" >> "$GITHUB_OUTPUT" + fi + # Caller workflow decides whether to fail the job on non-zero rc. + return 0 +} + +cmd_collect_logs() { + require_env SSH_CONFIG_FILE + require_env REMOTE_WORK_DIR + mkdir -p ./reports + if [ ! -f "$SSH_CONFIG_FILE" ]; then + echo "::warning::SSH config not present; skipping log collection." + return 0 + fi + echo "Copying logs from rvs-target:${REMOTE_WORK_DIR}/reports/ -> ./reports/" + scp -q -F "$SSH_CONFIG_FILE" "rvs-target:${REMOTE_WORK_DIR}/reports/*.log" ./reports/ || \ + echo "::warning::No log files retrieved from target node." + ls -la ./reports/ || true +} + +cmd_capture_versions() { + require_env SSH_CONFIG_FILE + require_env RVS_BIN + require_env INSTALL_DIR + require_env TARGET_ROCM_PATH + + local llvm_host_rt + llvm_host_rt=$(rvs_llvm_host_runtime_dir) + local rvs_version target_rocm_version + rvs_version=$( + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "RVS_BIN='$RVS_BIN' INSTALL_DIR='$INSTALL_DIR' TARGET_ROCM_PATH='$TARGET_ROCM_PATH' RVS_LLVM_HOST_RUNTIME_DIR='$llvm_host_rt' bash -s" <<'REMOTE' 2>/dev/null | head -1 || echo "unknown" +export LD_LIBRARY_PATH="${RVS_LLVM_HOST_RUNTIME_DIR:+$RVS_LLVM_HOST_RUNTIME_DIR:}${INSTALL_DIR}/lib:${TARGET_ROCM_PATH}/lib/rocm_sysdeps/lib:${TARGET_ROCM_PATH}/lib/llvm/lib:${TARGET_ROCM_PATH}/lib:${LD_LIBRARY_PATH:-}" +"$RVS_BIN" --version 2>/dev/null +REMOTE + ) + target_rocm_version=$( + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "TARGET_ROCM_PATH='$TARGET_ROCM_PATH' bash -s" <<'REMOTE' 2>/dev/null || echo "unknown" +cat "${TARGET_ROCM_PATH}/.info/version" 2>/dev/null \ + || cat "${TARGET_ROCM_PATH}/share/doc/rocm-core/version" 2>/dev/null \ + || echo "unknown" +REMOTE + ) + + if [ -n "${GITHUB_OUTPUT:-}" ]; then + echo "rvs_version=$rvs_version" >> "$GITHUB_OUTPUT" + echo "target_rocm_version=$target_rocm_version" >> "$GITHUB_OUTPUT" + fi +} + +load_pkg_meta_from_artifact() { + if [ -f ./pkg-meta/tarball_name ] && [ -z "${TARBALL_NAME:-}" ]; then + TARBALL_NAME="$(tr -d '\r' < ./pkg-meta/tarball_name)" + export TARBALL_NAME + fi + if [ -f ./pkg-meta/tarball_url ] && [ -z "${TARBALL_URL:-}" ]; then + TARBALL_URL="$(tr -d '\r' < ./pkg-meta/tarball_url)" + export TARBALL_URL + fi +} + +cmd_build_report() { + load_pkg_meta_from_artifact + # Job outputs sometimes drop tarball_name when tarball_url is multiline; recover from URL. + if [ -z "${TARBALL_NAME:-}" ] && [ -n "${TARBALL_URL:-}" ]; then + TARBALL_NAME="$(basename "${TARBALL_URL%%\?*}")" + export TARBALL_NAME + fi + require_env TARBALL_NAME + require_env TARGET_ROCM_PATH + require_env REMOTE_WORK_DIR + require_env RVS_BIN + + # TARBALL_URL is optional: the report job reads it from install-rvs-on-target outputs. + # If that output was empty (e.g. legacy echo-to-GITHUB_OUTPUT with special URL chars), + # still emit SUMMARY.md with the tarball filename. + local tarball_url_display="${TARBALL_URL:-}" + if [ -z "$tarball_url_display" ]; then + tarball_url_display="_(unavailable — check install job resolve step / use heredoc GITHUB_OUTPUT for URLs with special characters)_" + fi + + local rc4="${RVS_LEVEL4_RC:-0}" + local start="${RVS_LEVEL4_START:-}" + local end="${RVS_LEVEL4_END:-}" + local rvs_version="${RVS_VERSION:-unknown}" + local target_rocm_version="${TARGET_ROCM_VERSION:-unknown}" + local run_id="${GITHUB_RUN_ID:-local}" + local server_url="${GITHUB_SERVER_URL:-https://github.com}" + local repository="${GITHUB_REPOSITORY:-local/repo}" + local event_name="${GITHUB_EVENT_NAME:-local}" + + mkdir -p ./reports + local s4 overall + if [ "$rc4" -eq 0 ]; then + s4=PASS + overall=PASS + else + s4=FAIL + overall=FAIL + fi + + local report_title="${REPORT_TITLE:-# RVS PR Tests Report}" + local pkg_label="Linux tarball" + if [[ "${TARBALL_NAME}" == *.deb ]]; then + pkg_label="Deb package" + fi + local artifact_hint="${REPORT_ARTIFACT_HINT:-rvs-pr-report-${run_id}}" + + local report=./reports/SUMMARY.md + { + echo "${report_title}" + echo "" + echo "| Field | Value |" + echo "|---|---|" + echo "| Run | [\`${run_id}\`](${server_url}/${repository}/actions/runs/${run_id}) |" + echo "| Trigger | \`${event_name}\` |" + echo "| Target ROCm path | \`${TARGET_ROCM_PATH}\` (version \`${target_rocm_version}\`) |" + echo "| Remote work dir | \`${REMOTE_WORK_DIR}\` |" + echo "| ${pkg_label} | \`${TARBALL_NAME}\` |" + echo "| Source URL | ${tarball_url_display} |" + echo "| RVS version | \`${rvs_version}\` |" + echo "| Overall result | **${overall}** |" + echo "" + echo "## Results" + echo "" + echo "| Test | Command | Result | Exit | Started (UTC) | Ended (UTC) |" + echo "|---|---|:--:|---:|---|---|" + echo "| Level 4 | \`${RVS_BIN} -r 4\` | ${s4} | ${rc4} | ${start} | ${end} |" + echo "" + echo "## Logs" + echo "" + echo "Full stdout/stderr from the level-4 run is attached to the artifact" + echo "\`${artifact_hint}\`:" + echo "" + echo "- \`rvs_level_4.log\`" + echo "- \`SUMMARY.md\` (this file)" + } > "$report" + + cat "$report" + if [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + cat "$report" >> "$GITHUB_STEP_SUMMARY" + fi + + if [ -n "${GITHUB_OUTPUT:-}" ]; then + echo "overall=$overall" >> "$GITHUB_OUTPUT" + echo "rc4=$rc4" >> "$GITHUB_OUTPUT" + fi +} + +cmd_cleanup_remote() { + set +e + if [ -z "${REMOTE_WORK_DIR:-}" ] || [ -z "${SSH_CONFIG_FILE:-}" ] || [ ! -f "$SSH_CONFIG_FILE" ]; then + return 0 + fi + ssh -q -F "$SSH_CONFIG_FILE" rvs-target "rm -rf '${REMOTE_WORK_DIR}'" || \ + echo "::warning::Failed to clean up ${REMOTE_WORK_DIR} on target node." +} + +cmd_cleanup_local_ssh() { + set +e + if [ -n "${SSH_KEY_FILE:-}" ]; then + rm -f "$SSH_KEY_FILE" + fi + if [ -n "${SSH_CONFIG_FILE:-}" ]; then + rm -f "$SSH_CONFIG_FILE" "${SSH_CONFIG_FILE}.known_hosts" + fi +} + +main() { + if [ $# -lt 1 ]; then + usage + exit 1 + fi + case "$1" in + resolve-package-url) cmd_resolve_package_url "${2:-}" ;; + peek-latest-artifact-key) cmd_peek_latest_artifact_key "${2:-}" ;; + validate-config) cmd_validate_config ;; + setup-ssh) cmd_setup_ssh ;; + verify-rocm) cmd_verify_rocm ;; + download-tarball) cmd_download_tarball ;; + copy-to-target) cmd_copy_to_target ;; + install-rvs) cmd_install_rvs ;; + verify-rvs-binary) cmd_verify_rvs_binary ;; + run-level4) cmd_run_level4 ;; + collect-logs) cmd_collect_logs ;; + capture-versions) cmd_capture_versions ;; + build-report) cmd_build_report ;; + cleanup-remote) cmd_cleanup_remote ;; + cleanup-local-ssh) cmd_cleanup_local_ssh ;; + -h|--help) usage ;; + *) + echo "::error::Unknown command: $1" >&2 + usage + exit 1 + ;; + esac +} + +main "$@" diff --git a/rvs_release_test.sh b/rvs_release_test.sh new file mode 100755 index 000000000..f24ca152f --- /dev/null +++ b/rvs_release_test.sh @@ -0,0 +1,493 @@ +#!/bin/bash +################################################################################ +# Remote RVS release-tarball install + test driver (rvs-release-tests.yml only). +# Independent from rvs_nightly_test.sh (nightly) and rvs_pr_test.sh (PR). +################################################################################ + +set -euo pipefail + +usage() { + cat <<'EOF' +Usage: rvs_release_test.sh + +Commands (workflow order): + validate-config Fail fast; emit prepare job outputs to GITHUB_OUTPUT + setup-ssh Write SSH key/config under RUNNER_TEMP and verify connectivity + verify-rocm Remote rocminfo + amd-smi on TARGET_ROCM_PATH + download-tarball curl tarball to ./pkg/ on the orchestrator runner only (never on SSH target) + copy-to-target scp tarball to REMOTE_WORK_DIR/pkg on target + install-rvs Extract tarball on target under INSTALL_DIR + verify-rvs-binary Remote ldd check on RVS_BIN + run-level5 Run rvs -r 5; write rc/start/end to GITHUB_OUTPUT if set + collect-logs scp remote logs to ./reports/ + capture-versions Write rvs_version and target_rocm_version to GITHUB_OUTPUT + build-report Write ./reports/SUMMARY.md from env + prior outputs + cleanup-remote rm -rf REMOTE_WORK_DIR on target + cleanup-local-ssh Remove workflow-scoped key/config files +EOF +} + +require_env() { + local name="$1" + if [ -z "${!name:-}" ]; then + echo "::error::Required environment variable $name is not set" >&2 + exit 1 + fi +} + + +# Per-target libomp dir on the SSH target: ${ROCM}/lib/llvm/lib/$(clang --print-target-triple) +# TARGET_ROCM_PATH is a path on the target node, not on this orchestrator runner. +rvs_llvm_host_runtime_dir() { + require_env TARGET_ROCM_PATH + require_env SSH_CONFIG_FILE + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "TARGET_ROCM_PATH='${TARGET_ROCM_PATH}' bash -s" <<'REMOTE' +set -euo pipefail +clang="" +for candidate in \ + "${TARGET_ROCM_PATH}/bin/amdclang++" \ + "${TARGET_ROCM_PATH}/lib/llvm/bin/amdclang++" \ + "${TARGET_ROCM_PATH}/lib/llvm/bin/clang++"; do + [ -x "$candidate" ] && clang="$candidate" && break +done +[ -n "$clang" ] || { + echo "::warning::No ROCm clang under ${TARGET_ROCM_PATH}; skipping per-target libomp path" >&2 + exit 0 +} +triple=$("$clang" --print-target-triple 2>/dev/null) || exit 0 +dir="${TARGET_ROCM_PATH}/lib/llvm/lib/${triple}" +[ -d "$dir" ] || { + echo "::warning::llvm runtime triple dir not found under ${dir}" >&2 + exit 0 +} +echo "::notice::llvm runtime triple dir found under ${dir}" >&2 +printf '%s\n' "$dir" +REMOTE +} + + +cmd_validate_config() { + require_env TARBALL_NAME + require_env TARGET_NODE + require_env TARGET_ROCM_PATH + + local input_remote="${INPUT_REMOTE_WORK_DIR:-}" + local var_remote="${VAR_REMOTE_WORK_DIR:-}" + local run_id="${GITHUB_RUN_ID:-local}" + + if [ -n "$input_remote" ]; then + REMOTE_WORK_DIR="$input_remote" + elif [ -n "$var_remote" ]; then + REMOTE_WORK_DIR="$var_remote" + else + REMOTE_WORK_DIR="/tmp/rvs-release-${run_id}" + fi + + if [[ "$TARBALL_NAME" =~ ^amdrocm([0-9]+)- ]]; then + ROCM_MAJOR="${BASH_REMATCH[1]}" + else + echo "::error::Cannot parse ROCm major version from tarball name: $TARBALL_NAME" >&2 + exit 1 + fi + + INSTALL_DIR="/opt/rocm/extras-${ROCM_MAJOR}" + RVS_BIN="${INSTALL_DIR}/bin/rvs" + + echo "SSH target : (omitted — use secrets RVS_RELEASE_TARGET_* / workflow inputs)" + echo "Target ROCm path : $TARGET_ROCM_PATH" + echo "Remote work dir : $REMOTE_WORK_DIR" + echo "ROCm major : $ROCM_MAJOR" + echo "Expected RVS binary : $RVS_BIN" + + if [ -n "${GITHUB_OUTPUT:-}" ]; then + { + echo "remote_work_dir=$REMOTE_WORK_DIR" + echo "rocm_major=$ROCM_MAJOR" + echo "install_dir=$INSTALL_DIR" + echo "rvs_bin=$RVS_BIN" + } >> "$GITHUB_OUTPUT" + fi + + export REMOTE_WORK_DIR ROCM_MAJOR INSTALL_DIR RVS_BIN +} + +ssh_state_paths() { + local base="${RUNNER_TEMP:-/tmp}" + SSH_KEY_FILE="${base}/rvs_release_target_key" + SSH_CONFIG_FILE="${base}/rvs_release_target_ssh_config" + mkdir -p "$base" +} + +cmd_setup_ssh() { + require_env TARGET_NODE + require_env REMOTE_WORK_DIR + + ssh_state_paths + + if [ -n "${GITHUB_ENV:-}" ]; then + { + echo "SSH_KEY_FILE=$SSH_KEY_FILE" + echo "SSH_CONFIG_FILE=$SSH_CONFIG_FILE" + } >> "$GITHUB_ENV" + fi + + if [ -z "${SSH_PRIVATE_KEY:-}" ]; then + echo "::error::SSH_PRIVATE_KEY is not set; cannot SSH to target node." >&2 + exit 1 + fi + + install -m 600 /dev/null "$SSH_KEY_FILE" + printf '%s\n' "$SSH_PRIVATE_KEY" > "$SSH_KEY_FILE" + chmod 600 "$SSH_KEY_FILE" + + local known_hosts="${SSH_CONFIG_FILE}.known_hosts" + : > "$known_hosts" + chmod 600 "$known_hosts" + ssh-keyscan -T 15 -H "$TARGET_NODE" >> "$known_hosts" 2>/dev/null || true + + cat > "$SSH_CONFIG_FILE" <&2 + exit 1 + fi + echo "::notice::SSH connectivity OK." + + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "mkdir -p '${REMOTE_WORK_DIR}/pkg' '${REMOTE_WORK_DIR}/reports'" +} + +cmd_verify_rocm() { + require_env SSH_CONFIG_FILE + require_env TARGET_ROCM_PATH + + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "TARGET_ROCM_PATH='$TARGET_ROCM_PATH' bash -s" <<'REMOTE' +set -euo pipefail +echo "=== System ===" +uname -srvmo +echo +echo "=== Target ROCm path: ${TARGET_ROCM_PATH} ===" +if [ ! -d "${TARGET_ROCM_PATH}" ]; then + echo "::error::${TARGET_ROCM_PATH} does not exist on the target node." + exit 1 +fi +echo "=== ${TARGET_ROCM_PATH}/bin/rocminfo ===" +"${TARGET_ROCM_PATH}/bin/rocminfo" +echo +echo "=== ${TARGET_ROCM_PATH}/bin/amd-smi version ===" +"${TARGET_ROCM_PATH}/bin/amd-smi" version +echo +echo "::notice::ROCm prerequisites OK on target node at ${TARGET_ROCM_PATH}" +REMOTE +} + +cmd_download_tarball() { + require_env TARBALL_URL + require_env TARBALL_NAME + mkdir -p ./pkg + echo "Downloading release tarball on orchestrator runner (target node is not used for this fetch):" + echo " $TARBALL_URL" + curl -fL --max-time 600 -o "./pkg/${TARBALL_NAME}" "${TARBALL_URL}" + ls -la ./pkg/ + file "./pkg/${TARBALL_NAME}" || true +} + +cmd_copy_to_target() { + require_env SSH_CONFIG_FILE + require_env TARBALL_NAME + require_env REMOTE_WORK_DIR + echo "Copying ./pkg/${TARBALL_NAME} -> rvs-target:${REMOTE_WORK_DIR}/pkg/" + scp -q -F "$SSH_CONFIG_FILE" "./pkg/${TARBALL_NAME}" \ + "rvs-target:${REMOTE_WORK_DIR}/pkg/${TARBALL_NAME}" + ssh -q -F "$SSH_CONFIG_FILE" rvs-target "ls -la '${REMOTE_WORK_DIR}/pkg/'" +} + +cmd_install_rvs() { + require_env SSH_CONFIG_FILE + require_env TARBALL_NAME + require_env REMOTE_WORK_DIR + require_env ROCM_MAJOR + require_env INSTALL_DIR + require_env RVS_BIN + require_env TARGET_ROCM_PATH + + local llvm_host_rt + llvm_host_rt=$(rvs_llvm_host_runtime_dir) + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "TARBALL_NAME='$TARBALL_NAME' REMOTE_WORK_DIR='$REMOTE_WORK_DIR' ROCM_MAJOR='$ROCM_MAJOR' INSTALL_DIR='$INSTALL_DIR' RVS_BIN='$RVS_BIN' TARGET_ROCM_PATH='$TARGET_ROCM_PATH' RVS_LLVM_HOST_RUNTIME_DIR='$llvm_host_rt' bash -s" <<'REMOTE' +set -euo pipefail +PKG="${REMOTE_WORK_DIR}/pkg/${TARBALL_NAME}" +echo "Target ROCm path : ${TARGET_ROCM_PATH}" +echo "Detected ROCm major version : ${ROCM_MAJOR}" +echo "Install target : ${INSTALL_DIR}" +echo "Expected RVS binary : ${RVS_BIN}" +if [ ! -f "$PKG" ]; then + echo "::error::Tarball not found on target node at $PKG" + exit 1 +fi +probe="$INSTALL_DIR" +while [ ! -d "$probe" ] && [ "$probe" != "/" ]; do probe=$(dirname "$probe"); done +if [ -w "$probe" ]; then + echo "Installing into $INSTALL_DIR without sudo (writable path)" + mkdir -p "$INSTALL_DIR" + tar -xzf "$PKG" -C "$INSTALL_DIR" +else + echo "Installing into $INSTALL_DIR via sudo -n (path not user-writable)" + sudo -n mkdir -p "$INSTALL_DIR" + sudo -n tar -xzf "$PKG" -C "$INSTALL_DIR" +fi +if [ ! -x "$RVS_BIN" ]; then + echo "::error::rvs binary not found or not executable at $RVS_BIN after install" + ls -la "$INSTALL_DIR/" || true + ls -la "$INSTALL_DIR/bin/" || true + exit 1 +fi +export LD_LIBRARY_PATH="${RVS_LLVM_HOST_RUNTIME_DIR:+$RVS_LLVM_HOST_RUNTIME_DIR:}${INSTALL_DIR}/lib:${TARGET_ROCM_PATH}/lib/rocm_sysdeps/lib:${TARGET_ROCM_PATH}/lib/llvm/lib:${TARGET_ROCM_PATH}/lib:${LD_LIBRARY_PATH:-}" +echo "Installed RVS at: $RVS_BIN" +"$RVS_BIN" --version || true +REMOTE +} + +cmd_verify_rvs_binary() { + require_env SSH_CONFIG_FILE + require_env RVS_BIN + require_env INSTALL_DIR + require_env TARGET_ROCM_PATH + + local llvm_host_rt + llvm_host_rt=$(rvs_llvm_host_runtime_dir) + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "RVS_BIN='$RVS_BIN' INSTALL_DIR='$INSTALL_DIR' TARGET_ROCM_PATH='$TARGET_ROCM_PATH' RVS_LLVM_HOST_RUNTIME_DIR='$llvm_host_rt' bash -s" <<'REMOTE' +set -euo pipefail +if [ ! -x "$RVS_BIN" ]; then + echo "::error::$RVS_BIN was not produced by tarball extraction on target." + exit 1 +fi +export LD_LIBRARY_PATH="${RVS_LLVM_HOST_RUNTIME_DIR:+$RVS_LLVM_HOST_RUNTIME_DIR:}${INSTALL_DIR}/lib:${TARGET_ROCM_PATH}/lib/rocm_sysdeps/lib:${TARGET_ROCM_PATH}/lib/llvm/lib:${TARGET_ROCM_PATH}/lib:${LD_LIBRARY_PATH:-}" +echo "=== ldd $RVS_BIN ===" +LDD_OUTPUT=$(ldd "$RVS_BIN" 2>&1 || true) +echo "$LDD_OUTPUT" +if echo "$LDD_OUTPUT" | grep -q "not found"; then + echo "::error::RVS binary has unresolved library dependencies on target (see above)." + exit 1 +fi +echo "::notice::RVS binary's library dependencies resolved OK on target." +REMOTE +} + +cmd_run_level5() { + require_env SSH_CONFIG_FILE + require_env RVS_BIN + require_env INSTALL_DIR + require_env REMOTE_WORK_DIR + require_env TARGET_ROCM_PATH + + set +e + local start end rc + start=$(date -u +%FT%TZ) + echo "::group::RVS level 5 (${RVS_BIN} -r 5) against ${TARGET_ROCM_PATH}" + local llvm_host_rt + llvm_host_rt=$(rvs_llvm_host_runtime_dir) + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "RVS_BIN='$RVS_BIN' INSTALL_DIR='$INSTALL_DIR' REMOTE_WORK_DIR='$REMOTE_WORK_DIR' TARGET_ROCM_PATH='$TARGET_ROCM_PATH' RVS_LLVM_HOST_RUNTIME_DIR='$llvm_host_rt' bash -s" <<'REMOTE' +set +e +export LD_LIBRARY_PATH="${RVS_LLVM_HOST_RUNTIME_DIR:+$RVS_LLVM_HOST_RUNTIME_DIR:}${INSTALL_DIR}/lib:${TARGET_ROCM_PATH}/lib/rocm_sysdeps/lib:${TARGET_ROCM_PATH}/lib/llvm/lib:${TARGET_ROCM_PATH}/lib:${LD_LIBRARY_PATH:-}" +mkdir -p "${REMOTE_WORK_DIR}/reports" +"$RVS_BIN" -r 5 2>&1 | tee "${REMOTE_WORK_DIR}/reports/rvs_level_5.log" +RC=${PIPESTATUS[0]} +echo "remote_rc=$RC" +exit $RC +REMOTE + rc=$? + echo "::endgroup::" + end=$(date -u +%FT%TZ) + + if [ -n "${GITHUB_OUTPUT:-}" ]; then + echo "rc=$rc" >> "$GITHUB_OUTPUT" + echo "start=$start" >> "$GITHUB_OUTPUT" + echo "end=$end" >> "$GITHUB_OUTPUT" + fi + return 0 +} + +cmd_collect_logs() { + require_env SSH_CONFIG_FILE + require_env REMOTE_WORK_DIR + mkdir -p ./reports + if [ ! -f "$SSH_CONFIG_FILE" ]; then + echo "::warning::SSH config not present; skipping log collection." + return 0 + fi + echo "Copying logs from rvs-target:${REMOTE_WORK_DIR}/reports/ -> ./reports/" + scp -q -F "$SSH_CONFIG_FILE" "rvs-target:${REMOTE_WORK_DIR}/reports/*.log" ./reports/ || \ + echo "::warning::No log files retrieved from target node." + ls -la ./reports/ || true +} + +cmd_capture_versions() { + require_env SSH_CONFIG_FILE + require_env RVS_BIN + require_env INSTALL_DIR + require_env TARGET_ROCM_PATH + + local llvm_host_rt + llvm_host_rt=$(rvs_llvm_host_runtime_dir) + local rvs_version target_rocm_version + rvs_version=$( + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "RVS_BIN='$RVS_BIN' INSTALL_DIR='$INSTALL_DIR' TARGET_ROCM_PATH='$TARGET_ROCM_PATH' RVS_LLVM_HOST_RUNTIME_DIR='$llvm_host_rt' bash -s" <<'REMOTE' 2>/dev/null | head -1 || echo "unknown" +export LD_LIBRARY_PATH="${RVS_LLVM_HOST_RUNTIME_DIR:+$RVS_LLVM_HOST_RUNTIME_DIR:}${INSTALL_DIR}/lib:${TARGET_ROCM_PATH}/lib/rocm_sysdeps/lib:${TARGET_ROCM_PATH}/lib/llvm/lib:${TARGET_ROCM_PATH}/lib:${LD_LIBRARY_PATH:-}" +"$RVS_BIN" --version 2>/dev/null +REMOTE + ) + target_rocm_version=$( + ssh -q -F "$SSH_CONFIG_FILE" rvs-target \ + "TARGET_ROCM_PATH='$TARGET_ROCM_PATH' bash -s" <<'REMOTE' 2>/dev/null || echo "unknown" +cat "${TARGET_ROCM_PATH}/.info/version" 2>/dev/null \ + || cat "${TARGET_ROCM_PATH}/share/doc/rocm-core/version" 2>/dev/null \ + || echo "unknown" +REMOTE + ) + + if [ -n "${GITHUB_OUTPUT:-}" ]; then + echo "rvs_version=$rvs_version" >> "$GITHUB_OUTPUT" + echo "target_rocm_version=$target_rocm_version" >> "$GITHUB_OUTPUT" + fi +} + +cmd_build_report() { + require_env TARBALL_NAME + require_env TARGET_ROCM_PATH + require_env REMOTE_WORK_DIR + require_env RVS_BIN + + local tarball_url_display="${TARBALL_URL:-}" + if [ -z "$tarball_url_display" ]; then + tarball_url_display="_(unavailable — check install job resolve step / use heredoc GITHUB_OUTPUT for URLs with special characters)_" + fi + + local rc5="${RVS_LEVEL5_RC:-0}" + local start="${RVS_LEVEL5_START:-}" + local end="${RVS_LEVEL5_END:-}" + local rvs_version="${RVS_VERSION:-unknown}" + local target_rocm_version="${TARGET_ROCM_VERSION:-unknown}" + local run_id="${GITHUB_RUN_ID:-local}" + local server_url="${GITHUB_SERVER_URL:-https://github.com}" + local repository="${GITHUB_REPOSITORY:-local/repo}" + local event_name="${GITHUB_EVENT_NAME:-local}" + + mkdir -p ./reports + local s5 overall + if [ "$rc5" -eq 0 ]; then + s5=PASS + overall=PASS + else + s5=FAIL + overall=FAIL + fi + + local report=./reports/SUMMARY.md + { + echo "# RVS Release Test Report" + echo "" + echo "| Field | Value |" + echo "|---|---|" + echo "| Run | [\`${run_id}\`](${server_url}/${repository}/actions/runs/${run_id}) |" + echo "| Trigger | \`${event_name}\` |" + echo "| Target ROCm path | \`${TARGET_ROCM_PATH}\` (version \`${target_rocm_version}\`) |" + echo "| Remote work dir | \`${REMOTE_WORK_DIR}\` |" + echo "| Tarball | \`${TARBALL_NAME}\` |" + echo "| Source URL | ${tarball_url_display} |" + echo "| RVS version | \`${rvs_version}\` |" + echo "| Overall result | **${overall}** |" + echo "" + echo "## Results" + echo "" + echo "| Test | Command | Result | Exit | Started (UTC) | Ended (UTC) |" + echo "|---|---|:--:|---:|---|---|" + echo "| Level 5 | \`${RVS_BIN} -r 5\` | ${s5} | ${rc5} | ${start} | ${end} |" + echo "" + echo "## Logs" + echo "" + echo "Full stdout/stderr from the level-5 run is attached to the artifact" + echo "\`rvs-release-report-${run_id}\`:" + echo "" + echo "- \`rvs_level_5.log\`" + echo "- \`SUMMARY.md\` (this file)" + } > "$report" + + cat "$report" + if [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + cat "$report" >> "$GITHUB_STEP_SUMMARY" + fi + + if [ -n "${GITHUB_OUTPUT:-}" ]; then + echo "overall=$overall" >> "$GITHUB_OUTPUT" + echo "rc5=$rc5" >> "$GITHUB_OUTPUT" + fi +} + +cmd_cleanup_remote() { + set +e + if [ -z "${REMOTE_WORK_DIR:-}" ] || [ -z "${SSH_CONFIG_FILE:-}" ] || [ ! -f "$SSH_CONFIG_FILE" ]; then + return 0 + fi + ssh -q -F "$SSH_CONFIG_FILE" rvs-target "rm -rf '${REMOTE_WORK_DIR}'" || \ + echo "::warning::Failed to clean up ${REMOTE_WORK_DIR} on target node." +} + +cmd_cleanup_local_ssh() { + set +e + if [ -n "${SSH_KEY_FILE:-}" ]; then + rm -f "$SSH_KEY_FILE" + fi + if [ -n "${SSH_CONFIG_FILE:-}" ]; then + rm -f "$SSH_CONFIG_FILE" "${SSH_CONFIG_FILE}.known_hosts" + fi +} + +main() { + if [ $# -lt 1 ]; then + usage + exit 1 + fi + case "$1" in + validate-config) cmd_validate_config ;; + setup-ssh) cmd_setup_ssh ;; + verify-rocm) cmd_verify_rocm ;; + download-tarball) cmd_download_tarball ;; + copy-to-target) cmd_copy_to_target ;; + install-rvs) cmd_install_rvs ;; + verify-rvs-binary) cmd_verify_rvs_binary ;; + run-level5) cmd_run_level5 ;; + collect-logs) cmd_collect_logs ;; + capture-versions) cmd_capture_versions ;; + build-report) cmd_build_report ;; + cleanup-remote) cmd_cleanup_remote ;; + cleanup-local-ssh) cmd_cleanup_local_ssh ;; + -h|--help) usage ;; + *) + echo "::error::Unknown command: $1" >&2 + usage + exit 1 + ;; + esac +} + +main "$@" diff --git a/rvslib/CMakeLists.txt b/rvslib/CMakeLists.txt index 79b1bc20b..8b7349eb2 100644 --- a/rvslib/CMakeLists.txt +++ b/rvslib/CMakeLists.txt @@ -43,7 +43,7 @@ add_compile_options(-Wall ) add_compile_options(-fPIC) add_compile_options(-DRVS_OS_TYPE_NUM=${RVS_OS_TYPE_NUM}) -find_package(OpenMP) + if (RVS_COVERAGE) add_compile_options(-o0 -fprofile-arcs -ftest-coverage) @@ -173,7 +173,12 @@ target_compile_definitions(${RVS_TARGET} PRIVATE ROCM_USE_FLOAT16) set_target_properties(${RVS_TARGET} PROPERTIES SUFFIX .so.${LIB_VERSION_STRING} LIBRARY_OUTPUT_DIRECTORY ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}) -target_link_libraries(${RVS_TARGET} ${AMD_SMI_LIB}) +target_compile_options(${RVS_TARGET} PRIVATE -fopenmp) +if(FETCH_ROCMPATH_FROM_ROCMCORE) + target_link_libraries(${RVS_TARGET} ${AMD_SMI_LIB} ${ROCM_CORE} -fopenmp) +else() + target_link_libraries(${RVS_TARGET} ${AMD_SMI_LIB} -fopenmp) +endif() ## Install shared library librvslib.so add_custom_command(TARGET ${RVS_TARGET} POST_BUILD COMMAND ln -fs ./lib${RVS}.so.${LIB_VERSION_STRING} lib${RVS}.so.${VERSION_MAJOR} WORKING_DIRECTORY ${CMAKE_RUNTIME_OUTPUT_DIRECTORY} diff --git a/smqt.so/.gitignore b/smqt.so/.gitignore deleted file mode 100644 index bfe6c3ed1..000000000 --- a/smqt.so/.gitignore +++ /dev/null @@ -1,9 +0,0 @@ -/.settings/ -/CMakeFiles/ -/Debug/ -/build/ -/cmake_install.cmake -/Makefile -/.project -/lib*.so.*.*.* - diff --git a/smqt.so/CMakeLists.txt b/smqt.so/CMakeLists.txt deleted file mode 100644 index 806422abf..000000000 --- a/smqt.so/CMakeLists.txt +++ /dev/null @@ -1,147 +0,0 @@ -################################################################################ -## -## Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. -## -## MIT LICENSE: -## Permission is hereby granted, free of charge, to any person obtaining a copy of -## this software and associated documentation files (the "Software"), to deal in -## the Software without restriction, including without limitation the rights to -## use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies -## of the Software, and to permit persons to whom the Software is furnished to do -## so, subject to the following conditions: -## -## The above copyright notice and this permission notice shall be included in all -## copies or substantial portions of the Software. -## -## THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -## IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -## FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -## AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -## LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -## OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE -## SOFTWARE. -## -################################################################################ - -cmake_minimum_required ( VERSION 3.5.0 ) -if ( ${CMAKE_BINARY_DIR} STREQUAL ${CMAKE_CURRENT_SOURCE_DIR}) - message(FATAL "In-source build is not allowed") -endif () -set (CMAKE_RUNTIME_OUTPUT_DIRECTORY "${CMAKE_BINARY_DIR}/bin") - -set ( RVS "smqt" ) -set ( RVS_PACKAGE "rvs-roct" ) -set ( RVS_COMPONENT "lib${RVS}" ) -set ( RVS_TARGET "${RVS}" ) - -project ( ${RVS_TARGET} ) - -message(STATUS "MODULE: ${RVS}") - -add_compile_options(-Wall ) -if (RVS_COVERAGE) - add_compile_options(-o0 -fprofile-arcs -ftest-coverage) - set(CMAKE_EXE_LINKER_FLAGS "--coverage") - set(CMAKE_SHARED_LINKER_FLAGS "--coverage") -endif() - -## Set default module path if not already set -if ( NOT DEFINED CMAKE_MODULE_PATH ) - set ( CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/../cmake_modules/" ) -endif () - -## Include common cmake modules -include ( utils ) - -## Setup the package version. -get_version ( "0.0.0" ) - -set ( BUILD_VERSION_MAJOR ${VERSION_MAJOR} ) -set ( BUILD_VERSION_MINOR ${VERSION_MINOR} ) -set ( BUILD_VERSION_PATCH ${VERSION_PATCH} ) -set ( LIB_VERSION_STRING "${BUILD_VERSION_MAJOR}.${BUILD_VERSION_MINOR}.${BUILD_VERSION_PATCH}" ) - -if ( DEFINED VERSION_BUILD AND NOT ${VERSION_BUILD} STREQUAL "" ) - set ( BUILD_VERSION_PATCH "${BUILD_VERSION_PATCH}-${VERSION_BUILD}" ) -endif () -set ( BUILD_VERSION_STRING "${BUILD_VERSION_MAJOR}.${BUILD_VERSION_MINOR}.${BUILD_VERSION_PATCH}" ) - -## make version numbers visible to C code -add_compile_options(-DBUILD_VERSION_MAJOR=${VERSION_MAJOR}) -add_compile_options(-DBUILD_VERSION_MINOR=${VERSION_MINOR}) -add_compile_options(-DBUILD_VERSION_PATCH=${VERSION_PATCH}) -add_compile_options(-DLIB_VERSION_STRING="${LIB_VERSION_STRING}") -add_compile_options(-DBUILD_VERSION_STRING="${BUILD_VERSION_STRING}") - -# Determine HSA_PATH -if(NOT DEFINED HIPCC_PATH) - if(NOT DEFINED ENV{HIPCC_PATH}) - set(HIPCC_PATH "${ROCM_PATH}" CACHE PATH "Path to which hipcc runtime has been installed") - else() - set(HIPCC_PATH $ENV{HIPCC_PATH} CACHE PATH "Path to which hipcc runtime has been installed") - endif() -endif() - -# Add HIP_VERSION to CMAKE__FLAGS -set(HIP_HCC_BUILD_FLAGS "${HIP_HCC_BUILD_FLAGS} -DHIP_VERSION_MAJOR=${HIP_VERSION_MAJOR} -DHIP_VERSION_MINOR=${HIP_VERSION_MINOR} -DHIP_VERSION_PATCH=${HIP_VERSION_GITDATE}") - -set(HIP_HCC_BUILD_FLAGS) -set(HIP_HCC_BUILD_FLAGS "${HIP_HCC_BUILD_FLAGS} -fPIC ${HCC_CXX_FLAGS} -I${HIP_INC_DIR} ${ASAN_CXX_FLAGS}") - -# Set compiler and compiler flags -set(CMAKE_CXX_COMPILER "${HIPCC_PATH}/bin/hipcc") -set(CMAKE_C_COMPILER "${HIPCC_PATH}/bin/hipcc") -set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} ${HIP_HCC_BUILD_FLAGS}") -set(CMAKE_C_FLAGS "${CMAKE_C_FLAGS} ${HIP_HCC_BUILD_FLAGS}") -set(CMAKE_EXE_LINKER_FLAGS "${CMAKE_EXE_LINKER_FLAGS} ${ASAN_LD_FLAGS}") -set(CMAKE_SHARED_LINKER_FLAGS "${CMAKE_SHARED_LINKER_FLAGS} ${ASAN_LD_FLAGS}") - -if(BUILD_ADDRESS_SANITIZER) - execute_process(COMMAND ${CMAKE_CXX_COMPILER} --print-file-name=libclang_rt.asan-x86_64.so - OUTPUT_VARIABLE ASAN_LIB_FULL_PATH) - get_filename_component(ASAN_LIB_PATH ${ASAN_LIB_FULL_PATH} DIRECTORY) -else() - set(ASAN_LIB_PATH "$ENV{LD_LIBRARY_PATH}") -endif() - -## define include directories -include_directories(./ ../ pci) -# Add directories to look for library files to link -link_directories(${RVS_LIB_DIR} ${ASAN_LIB_PATH} ${ROCR_LIB_DIR}) -## additional libraries -set (PROJECT_LINK_LIBS rvslib libpci.so libm.so) - -## define source files -set(SOURCES src/rvs_module.cpp src/action.cpp) - -## define unit testing specific source files (mocking) -set(SOURCES_UT - test/action.cpp - ) - -## define target -add_library(${RVS_TARGET} SHARED ${SOURCES}) -set_target_properties(${RVS_TARGET} PROPERTIES - SUFFIX .so.${LIB_VERSION_STRING} - LIBRARY_OUTPUT_DIRECTORY ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}) -target_link_libraries(${RVS_TARGET} ${PROJECT_LINK_LIBS}) -add_dependencies(${RVS_TARGET} rvslib) - -add_custom_command(TARGET ${RVS_TARGET} POST_BUILD -COMMAND ln -fs ./lib${RVS}.so.${LIB_VERSION_STRING} lib${RVS}.so.${VERSION_MAJOR} WORKING_DIRECTORY ${CMAKE_RUNTIME_OUTPUT_DIRECTORY} -COMMAND ln -fs ./lib${RVS}.so.${VERSION_MAJOR} lib${RVS}.so WORKING_DIRECTORY ${CMAKE_RUNTIME_OUTPUT_DIRECTORY} -) - -install(TARGETS ${RVS_TARGET} LIBRARY DESTINATION ${CPACK_PACKAGING_INSTALL_PREFIX}/${CMAKE_INSTALL_LIBDIR}/rvs COMPONENT rvsmodule) -install(FILES "${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/lib${RVS}.so.${VERSION_MAJOR}" - DESTINATION ${CPACK_PACKAGING_INSTALL_PREFIX}/${CMAKE_INSTALL_LIBDIR}/rvs COMPONENT rvsmodule) -install(FILES "${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/lib${RVS}.so" - DESTINATION ${CPACK_PACKAGING_INSTALL_PREFIX}/${CMAKE_INSTALL_LIBDIR}/rvs COMPONENT rvsmodule) - -# TEST SECTION -if (RVS_BUILD_TESTS) - add_custom_command(TARGET ${RVS_TARGET} POST_BUILD - COMMAND ln -fs ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/lib${RVS}.so.${VERSION_MAJOR} ${RVS_BINTEST_FOLDER}/lib${RVS}.so WORKING_DIRECTORY ${CMAKE_RUNTIME_OUTPUT_DIRECTORY} - ) - include(${CMAKE_CURRENT_SOURCE_DIR}/tests.cmake) -endif() diff --git a/smqt.so/include/action.h b/smqt.so/include/action.h deleted file mode 100644 index ae6e8cf63..000000000 --- a/smqt.so/include/action.h +++ /dev/null @@ -1,80 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#ifndef SMQT_SO_INCLUDE_ACTION_H_ -#define SMQT_SO_INCLUDE_ACTION_H_ - -#include -#include "include/rvsactionbase.h" - -/** - * @class smqt_action - * @ingroup SMQT - * - * @brief SMQT action implementation class - * - * Derives from rvs::actionbase and implements actual action functionality - * in its run() method. - * - */ -class smqt_action : public rvs::actionbase { - public: - smqt_action(); - virtual ~smqt_action(); - virtual int run(void); - - private: - ulong get_property(std::string); - std::string pretty_print(ulong, uint16_t, std::string, std::string); - bool get_all_common_config_keys() override; - bool get_all_smqt_config_keys(); - std::string action_name; - - protected: - //! specified device_id - uint16_t dev_id; - //! actual BAR1 size - ulong bar1_size; - //! actual BAR2 size - ulong bar2_size; - //! actual BAR4 size - ulong bar4_size; - //! actual BAR5 size - ulong bar5_size; - //! actual BAR1 address - ulong bar1_base_addr; - //! actual BAR2 address - ulong bar2_base_addr; - //! actual BAR4 address - ulong bar4_base_addr; - -#ifdef RVS_UNIT_TEST - - protected: - virtual void on_set_device_gpu_id() = 0; - virtual void on_bar_data_read() = 0; -#endif -}; - -#endif // SMQT_SO_INCLUDE_ACTION_H_ diff --git a/smqt.so/include/rvs_module.h b/smqt.so/include/rvs_module.h deleted file mode 100644 index 7c14247e5..000000000 --- a/smqt.so/include/rvs_module.h +++ /dev/null @@ -1,30 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#ifndef SMQT_SO_INCLUDE_RVS_MODULE_H_ -#define SMQT_SO_INCLUDE_RVS_MODULE_H_ - -#include "include/rvsliblog.h" - -#endif /* SMQT_SO_INCLUDE_RVS_MODULE_H_ */ diff --git a/smqt.so/src/action.cpp b/smqt.so/src/action.cpp deleted file mode 100644 index 0d212c6ab..000000000 --- a/smqt.so/src/action.cpp +++ /dev/null @@ -1,382 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#include "include/action.h" - -#include -#include -#include -#include -#include -#include -#include -#ifdef __cplusplus -extern "C" { - #endif - #include - #include - #ifdef __cplusplus -} -#endif - -#include "include/rvs_key_def.h" -#include "include/rvs_module.h" -#include "include/pci_caps.h" -#include "include/gpu_util.h" -#include "include/rvsloglp.h" - -static constexpr auto MODULE_NAME = "smqt"; -static constexpr auto MODULE_NAME_CAPS = "SMQT"; - -using std::string; -using std::vector; -using std::cerr; -using std::cout; -using std::endl; - - -// config -ulong bar1_req_size, bar1_base_addr_min, bar1_base_addr_max; -ulong bar2_req_size, bar2_base_addr_min, bar2_base_addr_max; -ulong bar4_req_size, bar4_base_addr_min, bar4_base_addr_max, bar5_req_size; -bool keysts = true; -// Prints to the provided buffer a nice number of bytes (KB, MB, GB, etc) -string smqt_action::pretty_print(ulong bytes, uint16_t gpu_id, - string action_name, string bar_name) { - std::string suffix[5] = { " B", " KB", " MB", " GB", " TB"}; - std::stringstream ss; - - uint s = 0; // which suffix to use - double count = bytes; - while (count >= 1024 && s < 5) { - s++; - count /= 1024; - } - ss << "[" << action_name << "] smqt " << gpu_id << " " << - bar_name << " " - << bytes << " (" << std::fixed << std::setprecision(2) << - count << suffix[s] << ")"; - - return ss.str(); -} - -smqt_action::smqt_action() { - module_name = MODULE_NAME; -} - -smqt_action::~smqt_action() { - property.clear(); -} - -/** - * @brief reads all common configuration keys from - * the module's properties collection - * @return true if no fatal error occured, false otherwise - */ -bool smqt_action::get_all_common_config_keys() { - string msg, sdevid, sdev; - - - // get the action name - if (property_get(RVS_CONF_NAME_KEY, &action_name)) { - rvs::lp::Err("Action name missing", MODULE_NAME); - keysts = false; - } - - // get property value (a list of gpu id) - if (int sts = property_get_device()) { - switch (sts) { - case 1: - msg = "Invalid 'device' key value."; - break; - case 2: - msg = "Missing 'device' key."; - break; - } - rvs::lp::Err(msg, MODULE_NAME, action_name); - keysts = false; - } - - // get the property value if provided - if (property_get_int(RVS_CONF_DEVICEID_KEY, - &property_device_id, 0u) != 0) { - msg = "Invalid 'deviceid' key value."; - rvs::lp::Err(msg, MODULE_NAME, action_name); - keysts = false; - } - - // get property value (a list of device indexes) - if (int sts = property_get_device_index()) { - switch (sts) { - case 1: - msg = "Invalid 'device_index' key value."; - break; - case 2: - msg = "Missing 'device_index' key."; - break; - } - // default set as true - property_device_index_all = true; - rvs::lp::Log(msg, rvs::loginfo); - } - - return keysts; -} - -#define SMQT_FETCH_AND_CHECK(bar) \ -err = property_get_int(#bar, & bar); \ -switch (err) { \ - case 1: msg = "Invalid #bar key"; \ - rvs::lp::Err(msg, MODULE_NAME, action_name); \ - return false; \ - case 2: msg = "Missing #bar key"; \ - rvs::lp::Err(msg, MODULE_NAME, action_name); \ - return false; \ -} - -bool smqt_action::get_all_smqt_config_keys() { - int err = 0; - std::string msg; - - SMQT_FETCH_AND_CHECK(bar1_req_size) - SMQT_FETCH_AND_CHECK(bar2_req_size) - SMQT_FETCH_AND_CHECK(bar4_req_size) - SMQT_FETCH_AND_CHECK(bar5_req_size) - SMQT_FETCH_AND_CHECK(bar1_base_addr_min) - SMQT_FETCH_AND_CHECK(bar2_base_addr_min) - SMQT_FETCH_AND_CHECK(bar4_base_addr_min) - SMQT_FETCH_AND_CHECK(bar1_base_addr_max) - SMQT_FETCH_AND_CHECK(bar2_base_addr_max) - SMQT_FETCH_AND_CHECK(bar4_base_addr_max) - - return true; -} -/** - * @brief Implements action functionality - * Check if the sizes and addresses of BARs match the given ones - * @return 0 - success, non-zero otherwise - * */ - -int smqt_action::run(void) { - bool global_pass = true; - string msg; - struct pci_access *pacc; - bool devid_found = false; - rvs::action_result_t action_result; - - if (!get_all_common_config_keys()) { - msg = "Couldn't fetch common config keys from the configuration file!"; - rvs::lp::Err(msg, MODULE_NAME, action_name); - - action_result.state = rvs::actionstate::ACTION_COMPLETED; - action_result.status = rvs::actionstatus::ACTION_FAILED; - action_result.output = msg; - action_callback(&action_result); - - return -1; - } - - if (!get_all_smqt_config_keys()) { - msg = "Couldn't fetch smqt config keys from the configuration file!"; - rvs::lp::Err(msg, MODULE_NAME, action_name); - - action_result.state = rvs::actionstate::ACTION_COMPLETED; - action_result.status = rvs::actionstatus::ACTION_FAILED; - action_result.output = msg; - action_callback(&action_result); - - return -1; - } - - // get the pci_access structure - pacc = pci_alloc(); - // initialize the PCI library - pci_init(pacc); - // get the list of devices - pci_scan_bus(pacc); - - struct pci_dev *dev; - dev = pacc->devices; - - // iterate over devices - for (dev = pacc->devices; dev; dev = dev->next) { - bool pass = true; - // fil in the info - pci_fill_info(dev, PCI_FILL_IDENT | PCI_FILL_BASES \ - | PCI_FILL_CLASS | PCI_FILL_EXT_CAPS | PCI_FILL_CAPS | PCI_FILL_PHYS_SLOT); - - // computes the actual dev's location_id (sysfs entry) - uint16_t dev_location_id = ((((uint16_t) (dev->bus)) << 8) - | ((uint16_t) (dev->dev)) << 3 | ((uint16_t) (dev->func)) ); - - uint16_t gpu_id; - // if not and AMD GPU just continue - if (rvs::gpulist::location2gpu(dev_location_id, &gpu_id)) - continue; - -#ifdef RVS_UNIT_TEST - on_set_device_gpu_id(); -#endif - - // filter by device id if needed - if (property_device_id > 0) { - rvs::gpulist::gpu2device(gpu_id, &dev_id); - if (property_device_id != dev_id) { - continue; - keysts = false; - } - } - - devid_found = true; - - // filter by list of devices if needed - if (!property_device_all) { - if (property_device.end() == - std::find(property_device.begin(), property_device.end(), gpu_id)) - continue; - } - - // get actual values - bar1_base_addr = dev->base_addr[0]; - bar1_size = dev->size[0]; - bar2_base_addr = dev->base_addr[2]; - bar2_size = dev->size[2]; - bar4_base_addr = dev->base_addr[5]; - bar4_size = dev->size[5]; - bar5_size = dev->rom_size; - -#ifdef RVS_UNIT_TEST - on_bar_data_read(); -#endif - - // check if values are as expected - if (bar1_base_addr < bar1_base_addr_min || - bar1_base_addr > bar1_base_addr_max) - pass = false; - if (bar2_base_addr < bar2_base_addr_min || - bar2_base_addr > bar2_base_addr_max) - pass = false; - if (bar4_base_addr < bar4_base_addr_min || - bar4_base_addr > bar4_base_addr_max) - pass = false; - - if (bar1_req_size > bar1_size || - bar2_req_size < bar2_size || - bar4_req_size < bar4_size || - bar5_req_size < bar5_size) - pass = false; - - // loginfo - unsigned int sec; - unsigned int usec; - rvs::lp::get_ticks(&sec, &usec); - string msgs1, msgs2, msgs4, msgs5, msga1, msga2, msga4, pmsg, str, pass_str; - char hex_value[30]; - - if (pass) - pass_str = "true"; - else - pass_str = "false"; - - // formating bar1 size for print - msgs1 = pretty_print(bar1_size, gpu_id, action_name, "bar1_size"); - - // formating bar2 size for print - msgs2 = pretty_print(bar2_size, gpu_id, action_name, "bar2_size"); - - // formating bar4 size for print - msgs4 = pretty_print(bar4_size, gpu_id, action_name, "bar4_size"); - - // formating bar5 size for print - msgs5 = pretty_print(bar5_size, gpu_id, action_name, "bar5_size"); - - snprintf(hex_value, sizeof(hex_value), "%lX", bar1_base_addr); - msga1 = "[" + action_name + "] " + " smqt " + std::to_string(gpu_id) + - " bar1_base_addr " + hex_value; - snprintf(hex_value, sizeof(hex_value), "%lX", bar2_base_addr); - msga2 = "[" + action_name + "] " + " smqt " + std::to_string(gpu_id) + - " bar2_base_addr " + hex_value; - snprintf(hex_value, sizeof(hex_value), "%lX", bar4_base_addr); - msga4 = "[" + action_name + "] " + " smqt " + std::to_string(gpu_id) + - " bar4_base_addr " + hex_value; - pmsg = "[" + action_name + "] " + " smqt " + std::to_string(gpu_id) + - " " +pass_str; - - void* r = rvs::lp::LogRecordCreate("SMQT", action_name.c_str(), - rvs::loginfo, sec, usec); - - void* res = rvs::lp::LogRecordCreate("SMQT", action_name.c_str(), - rvs::logresults, sec, usec); - - rvs::lp::Log(msgs1, rvs::loginfo, sec, usec); - rvs::lp::Log(msga1, rvs::loginfo, sec, usec); - rvs::lp::Log(msgs2, rvs::loginfo, sec, usec); - rvs::lp::Log(msga2, rvs::loginfo, sec, usec); - rvs::lp::Log(msgs4, rvs::loginfo, sec, usec); - rvs::lp::Log(msga4, rvs::loginfo, sec, usec); - rvs::lp::Log(msgs5, rvs::loginfo, sec, usec); - rvs::lp::Log(pmsg, rvs::logresults); - - - action_result.state = rvs::actionstate::ACTION_RUNNING; - action_result.status = rvs::actionstatus::ACTION_SUCCESS; - action_result.output = msg.c_str(); - action_callback(&action_result); - - rvs::lp::AddInt(r, "gpu", gpu_id); - rvs::lp::AddString(r, "bar1_size", std::to_string(bar1_size)); - rvs::lp::AddString(r, "bar1_base_addr", std::to_string(bar1_base_addr)); - rvs::lp::AddString(r, "bar2_size", std::to_string(bar2_size)); - rvs::lp::AddString(r, "bar2_base_addr", std::to_string(bar2_base_addr)); - rvs::lp::AddString(r, "bar4_size", std::to_string(bar4_size)); - rvs::lp::AddString(r, "bar4_base_addr", std::to_string(bar4_base_addr)); - rvs::lp::AddString(r, "bar5_size", std::to_string(bar4_size)); - rvs::lp::AddString(res, "pass", std::to_string(pass)); - rvs::lp::LogRecordFlush(r); - rvs::lp::LogRecordFlush(res); - if (!pass) - global_pass = false; - } - if (!devid_found) { - global_pass = false; - msg = "No devices match criteria from the test configuration."; - rvs::lp::Err(msg, MODULE_NAME, action_name); - - action_result.state = rvs::actionstate::ACTION_COMPLETED; - action_result.status = rvs::actionstatus::ACTION_FAILED; - action_result.output = msg; - action_callback(&action_result); - - return -1; - } - - - action_result.state = rvs::actionstate::ACTION_COMPLETED; - action_result.status = (global_pass) ? rvs::actionstatus::ACTION_SUCCESS : rvs::actionstatus::ACTION_FAILED; - action_result.output = "SMQT Module action " + action_name + " completed"; - action_callback(&action_result); - - return global_pass ? 0 : -1; -} - diff --git a/smqt.so/src/rvs_module.cpp b/smqt.so/src/rvs_module.cpp deleted file mode 100644 index 7a1fee51d..000000000 --- a/smqt.so/src/rvs_module.cpp +++ /dev/null @@ -1,101 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#include "include/rvs_module.h" -#include "include/action.h" -#include "include/rvsloglp.h" -#include "include/gpu_util.h" - -/** - * @defgroup SMQT SMQT Module - * - * @brief SBIOS Mapping Qualification Tool - * - * The GPU SBIOS mapping qualification tool is designed to verify that a - * platform’s SBIOS has satisfied the BAR mapping requirements for VDI and Radeon - * Instinct products for ROCm support. - */ - -extern "C" int rvs_module_has_interface(int iid) { - int sts = 0; - switch (iid) { - case 0: - case 1: - sts = 1; - } - return sts; -} - -extern "C" const char* rvs_module_get_description(void) { - return (const char*)"The GPU SBIOS mapping qualification tool is designed to verify that a platform’s SBIOS has satisfied the BAR mapping requirements \n\tfor VDI and Radeon Instinct products for ROCm support."; -} - -extern "C" const char* rvs_module_get_config(void) { - return (const char*)"bar1_req_size(integer), bar1_base_addr_min(integer), " - "bar1_base_addr_max(integer), bar2_req_size(integer),\n\tbar2_base_addr_min(" - "integer), bar2_base_addr_max(integer), bar4_req_size(integer), " - "bar4_base_addr_min(integer), bar4_base_addr_max(integer),\n\t" - "bar5_req_size(integer)"; -} - -extern "C" const char* rvs_module_get_output(void) { - return (const char*)"bar1_size(integer), bar1_base_addr(integer), bar2_size" - "(integer), bar2_base_addr(integer), bar4_size(integer), bar4_base_addr" - "(integer),\n\tbar5_size(integer), pass(bool)"; -} - -extern "C" int rvs_module_init(void* pMi) { - rvs::lp::Initialize(static_cast(pMi)); - rvs::gpulist::Initialize(); - return 0; -} - -extern "C" int rvs_module_terminate(void) { - amdsmi_shut_down(); - return 0; -} - -extern "C" void* rvs_module_action_create(void) { - return static_cast(new smqt_action); -} - -extern "C" int rvs_module_action_destroy(void* pAction) { - delete static_cast(pAction); - return 0; -} - -extern "C" int rvs_module_action_property_set\ -(void* pAction, const char* Key, const char* Val) { - return static_cast(pAction)->property_set(Key, Val); -} - -extern "C" int rvs_module_action_callback_set(void* pAction, - rvs::callback_t callback, - void * user_param) { - return static_cast(pAction)->callback_set(callback, user_param); -} - -extern "C" int rvs_module_action_run(void* pAction) { - return static_cast(pAction)->run(); -} diff --git a/smqt.so/test/test_1.cpp b/smqt.so/test/test_1.cpp deleted file mode 100644 index bfb598fac..000000000 --- a/smqt.so/test/test_1.cpp +++ /dev/null @@ -1,46 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#include "gtest/gtest.h" -#include "include/action.h" -#include "test/unitsmqt.h" - -TEST(smqt, action) { - bar_data* bd = new bar_data; - bd->on_set_device_gpu_id(); - EXPECT_EQ(bd->get_dev_id(), 123); - bd->on_bar_data_read(); - ulong bar1_size, bar2_size, bar4_size, bar5_size; - ulong bar1_base_addr, bar2_base_addr, bar4_base_addr; - std::tie(bar1_size, bar2_size, bar4_size, bar5_size) = bd->get_bar_sizes(); - std::tie(bar1_base_addr, bar2_base_addr, bar4_base_addr) = bd->get_bar_addr(); - EXPECT_EQ(bar1_size, 2UL); - EXPECT_EQ(bar2_size, 3UL); - EXPECT_EQ(bar4_size, 5UL); - EXPECT_EQ(bar5_size, 4UL); - EXPECT_EQ(bar1_base_addr, 1UL); - EXPECT_EQ(bar2_base_addr, 6UL); - EXPECT_EQ(bar4_base_addr, 7UL); - delete bd; -} diff --git a/smqt.so/test/unitsmqt.cpp b/smqt.so/test/unitsmqt.cpp deleted file mode 100644 index 7fb756453..000000000 --- a/smqt.so/test/unitsmqt.cpp +++ /dev/null @@ -1,56 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#include "test/unitsmqt.h" -#include - -//! Default constructor -bar_data::bar_data() { -} - -//! Default destructor -bar_data::~bar_data() { - property.clear(); -} -void bar_data::on_set_device_gpu_id() { - dev_id = 123; -} -void bar_data::on_bar_data_read() { - bar1_size = 2; - bar2_size = 3; - bar4_size = 5; - bar5_size = 4; - bar1_base_addr = 1; - bar2_base_addr = 6; - bar4_base_addr = 7; -} -std::tuple bar_data::get_bar_sizes() { - return std::make_tuple(bar1_size, bar2_size, bar4_size, bar5_size); -} -std::tuple bar_data::get_bar_addr() { - return std::make_tuple(bar1_base_addr, bar2_base_addr, bar4_base_addr); -} -int bar_data::get_dev_id() { - return dev_id; -} diff --git a/smqt.so/test/unitsmqt.h b/smqt.so/test/unitsmqt.h deleted file mode 100644 index 5db878022..000000000 --- a/smqt.so/test/unitsmqt.h +++ /dev/null @@ -1,42 +0,0 @@ -/******************************************************************************** - * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. - * - * MIT LICENSE: - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to - * use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies - * of the Software, and to permit persons to whom the Software is furnished to do - * so, subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, - * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE - * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER - * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, - * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE - * SOFTWARE. - * - *******************************************************************************/ -#ifndef SMQT_SO_TEST_UNITSMQT_H_ -#define SMQT_SO_TEST_UNITSMQT_H_ -#include -#include "include/action.h" - - -class bar_data :public smqt_action { - public: - bar_data(); - virtual ~bar_data(); - virtual void on_set_device_gpu_id(); - virtual void on_bar_data_read(); - std::tuple get_bar_sizes(); - std::tuple get_bar_addr(); - int get_dev_id(); -}; - -#endif // SMQT_SO_TEST_UNITSMQT_H_ diff --git a/smqt.so/tests.cmake b/smqt.so/tests.cmake deleted file mode 100644 index 1c58c789c..000000000 --- a/smqt.so/tests.cmake +++ /dev/null @@ -1,49 +0,0 @@ -################################################################################ -## -## Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. -## -## MIT LICENSE: -## Permission is hereby granted, free of charge, to any person obtaining a copy of -## this software and associated documentation files (the "Software"), to deal in -## the Software without restriction, including without limitation the rights to -## use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies -## of the Software, and to permit persons to whom the Software is furnished to do -## so, subject to the following conditions: -## -## The above copyright notice and this permission notice shall be included in all -## copies or substantial portions of the Software. -## -## THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -## IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -## FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -## AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -## LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -## OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE -## SOFTWARE. -## -################################################################################ - -set(ROCBLAS_LIB "rocblas") -set(HIPRAND_LIB "hiprand") -set(HIPBLASLT_LIB "hipblaslt") -set(CORE_RUNTIME_NAME "hsa-runtime") -set(CORE_RUNTIME_TARGET "${CORE_RUNTIME_NAME}64") - -find_package(OpenMP) - -set(UT_LINK_LIBS libpthread.so libpci.so libm.so libdl.so ${AMD_SMI_LIB} OpenMP::OpenMP_CXX - ${ROCBLAS_LIB} ${CORE_RUNTIME_TARGET} ${ROCM_CORE} ${YAML_CPP_LIBRARIES} ${HIPRAND_LIB} ${HIPBLASLT_LIB} -) - -# Add directories to look for library files to link -link_directories(${ROCM_SMI_LIB_DIR} ${ROCBLAS_LIB_DIR} ${HIPRAND_LIB_DIR} ${HIPBLASLT_LIB_DIR} ${YAML_CPP_LIBRARY_DIR}) - -set (UT_SOURCES src/action.cpp test/unitsmqt.cpp -) - -# add unit tests -include(tests_unit) - -# Add configuration tests -include(tests_conf_logging) - diff --git a/src/gpu_util.cpp b/src/gpu_util.cpp index 9445f4e36..669c597dc 100644 --- a/src/gpu_util.cpp +++ b/src/gpu_util.cpp @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2025 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -48,6 +48,33 @@ std::vector rvs::gpulist::domain_id; std::map , uint16_t> rvs::gpulist::domain_loc_map; std::vector rvs::gpulist::pci_bdf; +const std::map gpu_dev_map = { + {0x74a1, "MI300X"}, // MI300X + {0x74a9, "MI300X-HF"}, // MI300-HF + {0x74a2, "MI308X"}, // MI308X + {0x74a8, "MI308X-HF"}, // MI308X-HF + {0x74a5, "MI325X"}, // MI325X + {0x75a0, "MI350X"}, // MI350X AC + {0x75a3, "MI355X"}, // MI355X LC + {0x75a8, "MI350P"}, // MI350P workstation (450W / 600W disambiguated by power cap) + {0x75c1, "MI450X"}, // MI450X + /* Navi 21 (RDNA2) - DID from internal PCI table */ + {0x73a3, "nv21"}, {0x73ae, "nv21"}, {0x73a5, "nv21"}, {0x73ab, "nv21"}, + {0x73af, "nv21"}, {0x73bf, "nv21"}, {0x73df, "nv21"}, {0x73e1, "nv21"}, + {0x73e3, "nv21"}, + /* Navi 31 (RDNA3) */ + {0x7448, "nv31"}, {0x7449, "nv31"}, {0x744a, "nv31"}, {0x744c, "nv31"}, + {0x7459, "nv31"}, {0x745e, "nv31"}, + /* Navi 32 (RDNA3) */ + {0x7460, "nv32"}, {0x7461, "nv32"}, {0x7470, "nv32"}, {0x7471, "nv32"}, + {0x7472, "nv32"}, {0x7478, "nv32"}, {0x747e, "nv32"}, {0x7480, "nv32"}, + {0x7481, "nv32"}, {0x7483, "nv32"}, + /* Navi 44 / gfx1200 (RDNA4) */ + {0x7590, "RX9060"}, + /* Navi 48 / gfx1201 (RDNA4) - default to RX9070; use -c for GRE/R9600D/gfx1201 */ + {0x7550, "RX9070"}, {0x7551, "RX9070"}, +}; + using std::vector; using std::string; using std::ifstream; @@ -430,9 +457,9 @@ void gpu_get_all_pci_bdf(std::vector& ppci_bdf) { } /** - * @brief Check if the GPU is die (chiplet) in Multi-Chip Module (MCM) GPU. + * @brief Check if the GPU is secondary die (chiplet) in Multi-Chip Module (MCM) GPU. * @param device_id GPU Device ID - * @return true if GPU is die in MCM GPU, false if GPU is single die GPU. + * @return true if GPU is secondary die in MCM GPU, false if GPU is single die GPU. **/ bool gpu_check_if_mcm_die (int idx) { amdsmi_status_t ret; @@ -495,6 +522,82 @@ int gpu_hip_to_smi_hdl(int hip_index, amdsmi_processor_handle* smi_hdl) { return -1; } +/** + * @brief Get node from hip index. + * @param hip_index GPU hip index + * @param node node id + * @return 0 if successful, -1 otherwise + **/ +int gpu_hip_to_node(int hip_index, int* node) { + int hip_num_gpu_devices = 0; + hipGetDeviceCount(&hip_num_gpu_devices); + if (hip_index >= hip_num_gpu_devices) { + return -1; + } + + char pciString[256] = {0}; + if (hipDeviceGetPCIBusId(pciString, 256, hip_index) != hipSuccess) { + return -1; + } + + uint64_t pDom = 0, pBus = 0, pDev = 0, pFun = 0; + if (sscanf(pciString, "%04x:%02x:%02x.%01x", + reinterpret_cast(&pDom), + reinterpret_cast(&pBus), + reinterpret_cast(&pDev), + reinterpret_cast(&pFun)) != 4) { + return -1; + } + + uint64_t hip_bdf_id = + ((((pDom) & 0x00000000ffffffff) << 32) | + (((pBus) & 0x00000000000000ff) << 8) | + (((pDev) & 0x000000000000001f) << 3) | + ((pFun) & 0x0000000000000007)); + + ifstream f_id, f_prop; + char path[KFD_PATH_MAX_LENGTH]; + int gpu_id; + std::string prop_name; + int num_nodes = gpu_num_subdirs(KFD_SYS_PATH_NODES, ""); + + for (int node_id = 0; node_id < num_nodes; node_id++) { + snprintf(path, KFD_PATH_MAX_LENGTH, "%s/%d/gpu_id", + KFD_SYS_PATH_NODES, node_id); + f_id.open(path); + f_id >> gpu_id; + f_id.close(); + + if (gpu_id == 0) continue; + + snprintf(path, KFD_PATH_MAX_LENGTH, "%s/%d/properties", + KFD_SYS_PATH_NODES, node_id); + f_prop.open(path); + + uint32_t domain_val = 0, location_val = 0; + while (f_prop >> prop_name) { + if (prop_name == "domain") { + f_prop >> domain_val; + } else if (prop_name == "location_id") { + f_prop >> location_val; + } else { + std::string dummy; + f_prop >> dummy; + } + } + f_prop.close(); + + uint64_t kfd_bdf_id = (((uint64_t)domain_val) << 32) | location_val; + + if (hip_bdf_id == kfd_bdf_id) { + *node = node_id; + return 0; + } + } + + return -1; +} + /** * @brief Initialize gpulist helper class * @return 0 if successful, -1 otherwise @@ -745,3 +848,87 @@ bool gpu_check_if_gpu_indexes (const std::vector &index) { return true; } +/** + * @brief Get GPU platform name + * @param void + * @return GPU plaform name if found, else null + **/ +std::string rvs::gpulist::gpu_get_platform_name (void) { + + uint16_t dev_id = 0; + + if (rvs::gpulist::device_id.size()) { + dev_id = rvs::gpulist::device_id[0]; + } + + for(auto i = 0; i < rvs::gpulist::device_id.size(); i++) { + + if(dev_id != rvs::gpulist::device_id[i]) { + return ""; + } + } + + auto it = gpu_dev_map.find(dev_id); + if (it == gpu_dev_map.end()) { + return ""; + } + + /* MI350P (0x75a8) shares one PCI device ID across the 450W and 600W SKUs; + differentiate by reading the hardware power cap reported by amdsmi. Each + known SKU is classified only when the reported cap lies within a +/-50 W + band around its nominal TDP (allowing for firmware/sysfs reporting drift); + an out-of-band value is treated as unknown and routed to the safe-default + fallback rather than silently bucketed to the nearest neighbor, so any + future intermediate SKU using the same device ID surfaces a warning. + Prefer max_power_cap from amdsmi_get_power_cap_info (the hardware limit, + in microwatts on Linux bare metal) since it is independent of any runtime + power-cap override the user may have set via rocm-smi which is in Watts. */ + if (it->second == "MI350P") { + constexpr uint64_t W_TO_UW = 1000000ULL; + + auto classify_uw = [](uint64_t cap_uw) -> std::string { + if (cap_uw >= 400 * W_TO_UW && cap_uw <= 500 * W_TO_UW) return "MI350P-450W"; + if (cap_uw >= 550 * W_TO_UW && cap_uw <= 650 * W_TO_UW) return "MI350P-600W"; + return ""; + }; + auto classify_w = [](uint32_t cap_w) -> std::string { + if (cap_w >= 400 && cap_w <= 500) return "MI350P-450W"; + if (cap_w >= 550 && cap_w <= 650) return "MI350P-600W"; + return ""; + }; + + auto smi_map = rvs::get_smi_pci_map(); + if (!smi_map.empty()) { + amdsmi_processor_handle hdl = smi_map.begin()->second; + amdsmi_power_cap_info_t cap_info{}; + if (amdsmi_get_power_cap_info(hdl, 0, &cap_info) == AMDSMI_STATUS_SUCCESS + && cap_info.max_power_cap > 0) { + std::string sku = classify_uw(cap_info.max_power_cap); + if (!sku.empty()) return sku; + std::cout << "MI350P max_power_cap " << (cap_info.max_power_cap / W_TO_UW) + << " W does not match any known SKU band (450 W / 600 W); " + "falling back to MI350P-450W. Use -c explicitly if this is " + "incorrect." << std::endl; + return "MI350P-450W"; + } + amdsmi_power_info_t pwr_info{}; + if (amdsmi_get_power_info(hdl, &pwr_info) == AMDSMI_STATUS_SUCCESS + && pwr_info.power_limit > 0) { + std::string sku = classify_w(pwr_info.power_limit); + if (!sku.empty()) return sku; + std::cout << "MI350P power_limit " << pwr_info.power_limit + << " W does not match any known SKU band (450 W / 600 W); " + "falling back to MI350P-450W. Use -c explicitly if this is " + "incorrect." << std::endl; + return "MI350P-450W"; + } + } + std::cout << "MI350P detected but power cap could not be read via amdsmi; " + "falling back to MI350P-450W. Use -c explicitly if this is " + "incorrect." << std::endl; + return "MI350P-450W"; + } + + return it->second; +} + diff --git a/src/pci_caps.cpp b/src/pci_caps.cpp index 75e3f8f3c..d833a7ff2 100644 --- a/src/pci_caps.cpp +++ b/src/pci_caps.cpp @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -81,7 +81,7 @@ unsigned int pci_dev_find_cap_offset(struct pci_dev *dev, unsigned char cap, * gets the max link speed * @param dev a pci_dev structure containing the PCI device information * @param buff pre-allocated char buffer - * @return + * @return */ void get_link_cap_max_speed(struct pci_dev *dev, char *buff) { const char *link_max_speed; @@ -129,7 +129,7 @@ void get_link_cap_max_speed(struct pci_dev *dev, char *buff) { * gets the PCI dev max link width * @param dev a pci_dev structure containing the PCI device information * @param buff pre-allocated char buffer - * @return + * @return */ void get_link_cap_max_width(struct pci_dev *dev, char *buff) { // get pci dev capabilities offset @@ -150,7 +150,7 @@ void get_link_cap_max_width(struct pci_dev *dev, char *buff) { * gets the current link speed * @param dev a pci_dev structure containing the PCI device information * @param buff pre-allocated char buffer - * @return + * @return */ void get_link_stat_cur_speed(struct pci_dev *dev, char *buff) { const char *link_cur_speed; @@ -198,7 +198,7 @@ void get_link_stat_cur_speed(struct pci_dev *dev, char *buff) { * gets the negotiated link width * @param dev a pci_dev structure containing the PCI device information * @param buff pre-allocated char buffer - * @return + * @return */ void get_link_stat_neg_width(struct pci_dev *dev, char *buff) { // get pci dev capabilities offset @@ -219,7 +219,7 @@ void get_link_stat_neg_width(struct pci_dev *dev, char *buff) { * gets the power limit value * @param dev a pci_dev structure containing the PCI device information * @param buff pre-allocated char buffer - * @return + * @return */ void get_slot_pwr_limit_value(struct pci_dev *dev, char *buff) { // get pci dev capabilities offset @@ -266,8 +266,8 @@ void get_slot_pwr_limit_value(struct pci_dev *dev, char *buff) { /** * gets PCI dev physical slot number * @param dev a pci_dev structure containing the PCI device information - * @param buff pre-allocated char buffer - * @return + * @param buff pre-allocated char buffer + * @return */ void get_slot_physical_num(struct pci_dev *dev, char *buff) { // get pci dev capabilities offset @@ -286,8 +286,8 @@ void get_slot_physical_num(struct pci_dev *dev, char *buff) { /** * gets PCI dev bus id * @param dev a pci_dev structure containing the PCI device information - * @param buff pre-allocated char buffer - * @return + * @param buff pre-allocated char buffer + * @return */ void get_pci_bus_id(struct pci_dev *dev, char *buff) { snprintf(buff, PCI_CAP_DATA_MAX_BUF_SIZE, "%0X", dev->bus); @@ -296,8 +296,8 @@ void get_pci_bus_id(struct pci_dev *dev, char *buff) { /** * gets PCI device id (it appears as a function just to keep compatibility with the array of pointer to function) * @param dev a pci_dev structure containing the PCI device information - * @param buff pre-allocated char buffer - * @return + * @param buff pre-allocated char buffer + * @return */ void get_device_id(struct pci_dev *dev, char *buff) { snprintf(buff, PCI_CAP_DATA_MAX_BUF_SIZE, "%u", dev->device_id); @@ -306,8 +306,8 @@ void get_device_id(struct pci_dev *dev, char *buff) { /** * gets PCI device's vendor id (it appears as a function just to keep compatibility with the array of pointer to function) * @param dev a pci_dev structure containing the PCI device information - * @param buff pre-allocated char buffer - * @return + * @param buff pre-allocated char buffer + * @return */ void get_vendor_id(struct pci_dev *dev, char *buff) { snprintf(buff, PCI_CAP_DATA_MAX_BUF_SIZE, "%u", dev->vendor_id); @@ -317,51 +317,39 @@ void get_vendor_id(struct pci_dev *dev, char *buff) { * gets the PCI dev driver name * @param dev a pci_dev structure containing the PCI device information * @param buf pre-allocated char buffer - * @return + * @return */ void get_kernel_driver(struct pci_dev *dev, char *buff) { - char name[1024], *drv, *base; + char name[256]; + char link_target[PCI_CAP_DATA_MAX_BUF_SIZE]; + char *drv; int n; - buff[0] = '\0'; - - //returning from here as there are other ways, this is - //not supported any longer - //the more simpler way is sudo dkms status snprintf(buff, PCI_CAP_DATA_MAX_BUF_SIZE, "%s", PCI_CAP_NOT_SUPPORTED); - if (dev->access == NULL) { - return; - } - - if (dev->access->method != PCI_ACCESS_SYS_BUS_PCI) { - return; - } - - base = pci_get_param(dev->access, const_cast("sysfs.path")); - if (!base || !base[0]) { + if (dev == NULL) { return; } - n = snprintf(name, sizeof(name), "%s/devices/%04x:%02x:%02x.%d/driver", - base, dev->domain, dev->bus, dev->dev, dev->func); + n = snprintf(name, sizeof(name), + "/sys/bus/pci/devices/%04x:%02x:%02x.%d/driver", + dev->domain, dev->bus, dev->dev, dev->func); if (n < 0 || n >= static_cast(sizeof(name))) { return; } - n = readlink(name, buff, PCI_CAP_DATA_MAX_BUF_SIZE); + n = readlink(name, link_target, PCI_CAP_DATA_MAX_BUF_SIZE - 1); if (n < 0) { return; } - if (n >= PCI_CAP_DATA_MAX_BUF_SIZE) { - return; - } - - buff[n] = 0; + link_target[n] = '\0'; - if ((drv = strrchr(buff, '/')) != NULL) + if ((drv = strrchr(link_target, '/')) != NULL) { snprintf(buff, PCI_CAP_DATA_MAX_BUF_SIZE, "%s", drv + 1); + } else { + snprintf(buff, PCI_CAP_DATA_MAX_BUF_SIZE, "%s", link_target); + } } /** @@ -405,39 +393,43 @@ void get_pwr_budgeting(struct pci_dev *dev, uint8_t pb_pm_state, snprintf(buff, PCI_CAP_DATA_MAX_BUF_SIZE, "%s", PCI_CAP_NOT_SUPPORTED); - if (cap_offset_pwbgd != 0) { - i = 0; - - do { - // Data select register size is 1 byte, it will select the DWORD(4 bytes) - // from budgeting data register corresponding to state. - // dont proceed if write to register fails - int rt = pci_write_byte(dev, cap_offset_pwbgd + PCI_PWR_DSR, i); - if(!rt){// 0 indicates error in writing - ++i; - continue; - } - w = pci_read_word(dev, cap_offset_pwbgd + PCI_PWR_DATA); - - if (!w) - return; - - pb_act_pm_state = PCI_PWR_DATA_PM_STATE(w); - pb_act_type = PCI_PWR_DATA_TYPE(w); - pb_act_power_rail = PCI_PWR_DATA_RAIL(w); - - if (pb_act_pm_state == pb_pm_state && pb_act_type == pb_type && - pb_act_power_rail == pb_power_rail) { - base = PCI_PWR_DATA_BASE(w); - scale = PCI_PWR_DATA_SCALE(w); - snprintf(buff, PCI_CAP_DATA_MAX_BUF_SIZE, "%.3fW", - base * pow(10, -scale)); - return; - } - - i++; - } while (i <= DSR_MAX_VAL); // DSR size is 1 byte, no need to run forever + if (cap_offset_pwbgd == 0 || dev->access == NULL) { + return; } + + i = 0; + + do { + // Data select register size is 1 byte, it will select the DWORD(4 bytes) + // from budgeting data register corresponding to state. + // dont proceed if write to register fails + int rt = pci_write_byte(dev, cap_offset_pwbgd + PCI_PWR_DSR, i); + if (!rt) { // 0 indicates error in writing + ++i; + continue; + } + w = pci_read_word(dev, cap_offset_pwbgd + PCI_PWR_DATA); + + if (!w) { + ++i; + continue; + } + + pb_act_pm_state = PCI_PWR_DATA_PM_STATE(w); + pb_act_type = PCI_PWR_DATA_TYPE(w); + pb_act_power_rail = PCI_PWR_DATA_RAIL(w); + + if (pb_act_pm_state == pb_pm_state && pb_act_type == pb_type && + pb_act_power_rail == pb_power_rail) { + base = PCI_PWR_DATA_BASE(w); + scale = PCI_PWR_DATA_SCALE(w); + snprintf(buff, PCI_CAP_DATA_MAX_BUF_SIZE, "%.3fW", + base * pow(10, -scale)); + return; + } + + i++; + } while (i <= DSR_MAX_VAL); // DSR size is 1 byte, no need to run forever } /** diff --git a/src/rvs_blas.cpp b/src/rvs_blas.cpp index 50ebc13d0..069e2ae6c 100644 --- a/src/rvs_blas.cpp +++ b/src/rvs_blas.cpp @@ -552,7 +552,9 @@ bool rvs_blas::copy_data_to_gpu(void) { return copy_data_to_gpu(); } - if(data_type == "fp4_r" || data_type == "fp6_e3m2_r" || data_type == "fp6_e2m3_r") { + if(data_type == "fp4_r" || data_type == "fp6_e3m2_r" || data_type == "fp6_e2m3_r" || + data_type == "mxfp8_e4m3_r" || data_type == "mxfp8_e5m2_r") { + if (blas_source == "hipblaslt") { return copy_data_to_gpu(); } @@ -656,7 +658,9 @@ bool rvs_blas::allocate_gpu_matrix_mem(void) { return allocate_gpu_matrix_mem(); } - if(data_type == "fp4_r" || data_type == "fp6_e3m2_r" || data_type == "fp6_e2m3_r") { + if(data_type == "fp4_r" || data_type == "fp6_e3m2_r" || data_type == "fp6_e2m3_r" || + data_type == "mxfp8_e4m3_r" || data_type == "mxfp8_e5m2_r") { + if (blas_source == "hipblaslt") { return allocate_gpu_matrix_mem(); } @@ -818,7 +822,8 @@ bool rvs_blas::allocate_host_matrix_mem(void) { hc = new rocblas_half[size_c]; } - if(data_type == "fp4_r" || data_type == "fp6_e3m2_r" || data_type == "fp6_e2m3_r") { + if(data_type == "fp4_r" || data_type == "fp6_e3m2_r" || data_type == "fp6_e2m3_r" || + data_type == "mxfp8_e4m3_r" || data_type == "mxfp8_e5m2_r") { if (blas_source == "hipblaslt") { ha = new uint8_t[size_a * block_count]; @@ -970,21 +975,13 @@ bool rvs_blas::is_gemm_op_complete(void) { * @brief performs the GEMM matrix multiplication operations * @return true if GPU was able to enqueue the GEMM operation, otherwise false */ -bool rvs_blas::run_blas_gemm(bool hot_call) { - - int calls = 0; +bool rvs_blas::run_blas_gemm(uint64_t num_calls) { if (is_error) return false; - /* Determine GEMM call iterations */ - if(true == hot_call) - calls = hot_calls; - else - calls = 1; - /* GEMM call iterations loop */ - for(int i = 0; i < calls; i++) { + for(uint64_t i = 0; i < num_calls; i++) { if(blas_source == "rocblas") { @@ -1230,9 +1227,31 @@ bool rvs_blas::run_blas_gemm(bool hot_call) { void * dc_p = (uint8_t *)dc + (i % block_count) * size_c * get_hipdatatype_size(hbl_out_datatype); void * dd_p = (uint8_t *)dd + (i % block_count) * size_d * get_hipdatatype_size(hbl_out_datatype); + double alpha_d, beta_d; + int32_t alpha_i, beta_i; + float alpha_f, beta_f; + const void *alpha_p, *beta_p; + + if (hbl_computetype == HIPBLAS_COMPUTE_64F) { + alpha_d = blas_alpha_val; + beta_d = blas_beta_val; + alpha_p = &alpha_d; + beta_p = &beta_d; + } else if (hbl_computetype == HIPBLAS_COMPUTE_32I) { + alpha_i = (int32_t)blas_alpha_val; + beta_i = (int32_t)blas_beta_val; + alpha_p = &alpha_i; + beta_p = &beta_i; + } else { + alpha_f = blas_alpha_val; + beta_f = blas_beta_val; + alpha_p = &alpha_f; + beta_p = &beta_f; + } + if (hipblasLtMatmul(hbl_handle, hbl_matmul[i % block_count], - &blas_alpha_val, da_p, hbl_layout_a, - db_p, hbl_layout_b, &blas_beta_val, + alpha_p, da_p, hbl_layout_a, + db_p, hbl_layout_b, beta_p, dc_p, hbl_layout_c, dd_p, hbl_layout_d, &hbl_heuristic_result.algo, hbl_workspace, hbl_heuristic_result.workspaceSize, @@ -1386,7 +1405,8 @@ void rvs_blas::generate_random_matrix_data(void) { } } - if(data_type == "fp4_r" || data_type == "fp6_e3m2_r" || data_type == "fp6_e2m3_r") { + if(data_type == "fp4_r" || data_type == "fp6_e3m2_r" || data_type == "fp6_e2m3_r" || + data_type == "mxfp8_e4m3_r" || data_type == "mxfp8_e5m2_r") { generateMXInput(hbl_datatype, ha, @@ -2508,7 +2528,7 @@ std::vector rvs_blas::generateMXInput(hipDataType dataType, isTranspose, isMatrixA); } - else if(static_cast(dataType) == HIP_R_6F_E2M3_EXT) + else if(static_cast(dataType) == HIP_R_6F_E2M3) { DGen::DataGenerator dgen; return generateData(dgen, @@ -2522,7 +2542,7 @@ std::vector rvs_blas::generateMXInput(hipDataType dataType, isTranspose, isMatrixA); } - else if(static_cast(dataType) == HIP_R_6F_E3M2_EXT) + else if(static_cast(dataType) == HIP_R_6F_E3M2) { DGen::DataGenerator dgen; return generateData(dgen, @@ -2536,7 +2556,7 @@ std::vector rvs_blas::generateMXInput(hipDataType dataType, isTranspose, isMatrixA); } - else if(static_cast(dataType) == HIP_R_4F_E2M1_EXT) + else if(static_cast(dataType) == HIP_R_4F_E2M1) { DGen::DataGenerator dgen; return generateData(dgen, diff --git a/src/rvs_util.cpp b/src/rvs_util.cpp index 042bf15d8..53f1e0b89 100644 --- a/src/rvs_util.cpp +++ b/src/rvs_util.cpp @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2025 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -25,9 +25,18 @@ #include "include/rvs_util.h" #include #include +#include +#include #include +#if defined(__linux__) +#include +#include +#endif #include #include +#ifdef FETCH_ROCMPATH_FROM_ROCMCORE +#include "rocm-core/rocm_getpath.h" +#endif #include "hip/hip_runtime.h" #include "hip/hip_runtime_api.h" #include "amd_smi/amdsmi.h" @@ -41,19 +50,20 @@ */ std::vector str_split(const std::string& str_val, - const std::string& delimiter) { - std::vector str_tokens; - int prev_pos = 0, cur_pos = 0; - do { - cur_pos = str_val.find(delimiter, prev_pos); - if (cur_pos == std::string::npos) - cur_pos = str_val.length(); - std::string token = str_val.substr(prev_pos, cur_pos - prev_pos); - if (!token.empty()) - str_tokens.push_back(token); - prev_pos = cur_pos + delimiter.length(); - } while (cur_pos < str_val.length() && prev_pos < str_val.length()); - return str_tokens; + const std::string& delimiter) { + + std::vector str_tokens; + int prev_pos = 0, cur_pos = 0; + do { + cur_pos = str_val.find(delimiter, prev_pos); + if (cur_pos == std::string::npos) + cur_pos = str_val.length(); + std::string token = str_val.substr(prev_pos, cur_pos - prev_pos); + if (!token.empty()) + str_tokens.push_back(token); + prev_pos = cur_pos + delimiter.length(); + } while (cur_pos < str_val.length() && prev_pos < str_val.length()); + return str_tokens; } @@ -63,12 +73,14 @@ std::vector str_split(const std::string& str_val, * @return true if std::string is a positive integer number, false otherwise */ bool is_positive_integer(const std::string& str_val) { - return !str_val.empty() - && std::find_if(str_val.begin(), str_val.end(), - [](char c) {return !std::isdigit(c);}) == str_val.end(); + + return !str_val.empty() + && std::find_if(str_val.begin(), str_val.end(), + [](char c) {return !std::isdigit(c);}) == str_val.end(); } int rvs_util_parse(const std::string& buff, bool* pval) { + if (buff.empty()) { // method empty return 2; // not found } @@ -87,25 +99,23 @@ int rvs_util_parse(const std::string& buff, bool* pval) { } void *json_node_create(std::string module_name, std::string action_name, - int log_level){ - unsigned int sec; - unsigned int usec; - - rvs::lp::get_ticks(&sec, &usec); - void *json_node = rvs::lp::LogRecordCreate(module_name.c_str(), - action_name.c_str(), log_level, sec, usec, true); - return json_node; -} + int log_level){ + unsigned int sec; + unsigned int usec; + rvs::lp::get_ticks(&sec, &usec); + void *json_node = rvs::lp::LogRecordCreate(module_name.c_str(), + action_name.c_str(), log_level, sec, usec, true); + return json_node; +} void *json_list_create(std::string lname, int log_level){ - void *json_node = rvs::lp::JsonNamedListCreate(lname.c_str() ,log_level); - return json_node; + void *json_node = rvs::lp::JsonNamedListCreate(lname.c_str() ,log_level); + return json_node; } - /** * summary: Fetches gpu id to index map for valid set of GPUs as per config. * Note: mcm_check is needed to output MCM specific messages while we iterate @@ -115,7 +125,8 @@ void *json_list_create(std::string lname, int log_level){ bool fetch_gpu_list(int hip_num_gpu_devices, map& gpus_device_index, const std::vector& property_device, const int& property_device_id, bool property_device_all, const std::vector& property_device_index, - bool property_device_index_all, bool mcm_check) { + bool property_device_index_all, bool mcm_check, + std::vector* mcm_type) { bool amd_gpus_found = false; bool mcm_die = false; @@ -180,7 +191,7 @@ bool fetch_gpu_list(int hip_num_gpu_devices, map& gpus_device_ind // if mcm check enabled, print message if device is MCM // MCM is only for mi200, so check only if gfx90a - if (mcm_check && (std::string{props.gcnArchName}.find("gfx90a") != std::string::npos) ){ + if (mcm_check && (std::string{props.gcnArchName}.find("gfx90a") != std::string::npos)){ std::stringstream msg_stream; mcm_die = gpu_check_if_mcm_die(i); if (mcm_die) { @@ -190,8 +201,16 @@ bool fetch_gpu_list(int hip_num_gpu_devices, map& gpus_device_ind rvs::lp::Log(msg_stream.str(), rvs::logresults); amd_mcm_gpu_found = true; + if (mcm_type) mcm_type->push_back(mcm_type_t::SECONDARY); + + } else { + if (mcm_type) mcm_type->push_back(mcm_type_t::PRIMARY); } + } else { + if(mcm_check && mcm_type) + mcm_type->push_back(mcm_type_t::PRIMARY); } + // excluding secondary die from test list, drops power reading substantially if (cur_gpu_selected) { // need not exclude secondary die, just log it out. gpus_device_index.insert @@ -214,47 +233,39 @@ bool fetch_gpu_list(int hip_num_gpu_devices, map& gpus_device_ind } void getBDF(int idx ,unsigned int& domain,unsigned int& bus,unsigned int& device,unsigned int& function){ - char pciString[256] = {0}; - auto hipRet = hipDeviceGetPCIBusId(pciString, 256, idx); - std::string msg; - if(hipRet != hipSuccess){ - msg = "For GPU:" + std::to_string(idx) + ", failed to get PCI Bus id"; - rvs::lp::Log(msg, rvs::logresults); - return; - } - if (sscanf(pciString, "%04x:%02x:%02x.%01x", reinterpret_cast(&domain), - reinterpret_cast(&bus), - reinterpret_cast(&device), - reinterpret_cast(&function)) != 4) { - msg = std::string("parsing incomplete for BDF id: ") + pciString ; - rvs::lp::Log(msg, rvs::logresults); - } -} -int display_gpu_info (void) { + char pciString[256] = {0}; + auto hipRet = hipDeviceGetPCIBusId(pciString, 256, idx); + std::string msg; + if(hipRet != hipSuccess){ + msg = "For GPU:" + std::to_string(idx) + ", failed to get PCI Bus id"; + rvs::lp::Log(msg, rvs::logresults); + return; + } + if (sscanf(pciString, "%04x:%02x:%02x.%01x", reinterpret_cast(&domain), + reinterpret_cast(&bus), + reinterpret_cast(&device), + reinterpret_cast(&function)) != 4) { + msg = std::string("parsing incomplete for BDF id: ") + pciString ; + rvs::lp::Log(msg, rvs::logresults); + } +} - struct device_info { - std::string bus; - std::string name; - int32_t node_id; - int32_t gpu_id; - int32_t device_id; - uint64_t bdfId;// this is pcie location id to uniquely identify device - }; +std::vector get_gpu_info (void) { char buf[256]; int hip_num_gpu_devices; - std::string errmsg = " No supported GPUs available."; std::vector gpu_info_list; hipGetDeviceCount(&hip_num_gpu_devices); if( hip_num_gpu_devices == 0){ - std::cout << std::endl << errmsg << std::endl; - return 0; + return {}; } + for (int i = 0; i < hip_num_gpu_devices; i++) { hipDeviceProp_t props; hipGetDeviceProperties(&props, i); + unsigned int pDom, pBus, pDev, pFun; getBDF(i , pDom, pBus, pDev, pFun); // compute device location_id (needed in order to identify this device @@ -295,6 +306,14 @@ int display_gpu_info (void) { std::sort(gpu_info_list.begin(), gpu_info_list.end(), [](const struct device_info& a, const struct device_info& b) { return a.bdfId < b.bdfId; }); + + return gpu_info_list; +} + +int display_gpu_info (std::vector gpu_info_list) { + + std::string errmsg = " No supported GPUs available."; + if (!gpu_info_list.empty()) { std::cout << "Supported GPUs available:\n"; for (const auto& info : gpu_info_list) { @@ -308,8 +327,8 @@ int display_gpu_info (void) { return 0; } - void json_add_primary_fields(std::string moduleName, std::string action_name){ + if (rvs::lp::JsonActionStartNodeCreate(moduleName.c_str(), action_name.c_str())){ rvs::lp::Err("json start create failed", moduleName, action_name); return; @@ -320,3 +339,271 @@ void cleanup_logs(){ rvs::lp::JsonEndNodeCreate(); } +std::string get_gpu_name (void) { + + rvs::gpulist::Initialize(); + + return rvs::gpulist::gpu_get_platform_name () ; +} + +std::string rvs_get_rocm_install_path_string(void) { +#ifdef FETCH_ROCMPATH_FROM_ROCMCORE + char* installPath = nullptr; + unsigned int installPathLen = 0; + PathErrors_t perr = getROCmInstallPath(&installPath, &installPathLen); + if (perr == PathSuccess && installPath != nullptr) { + std::string s(installPath); + free(installPath); + if (!s.empty()) { + return s; + } + } else if (installPath != nullptr) { + free(installPath); + } +#endif + if (const char* env = std::getenv("ROCM_PATH")) { + if (env[0] != '\0') { + return std::string(env); + } + } + return std::string(ROCM_PATH); +} + +namespace { + +std::string rvs_strip_trailing_slashes(std::string s) { + while (s.size() > 1 && s.back() == '/') { + s.pop_back(); + } + return s; +} + +std::string rvs_path_dirname(const std::string& path) { + if (path.empty()) { + return path; + } + size_t end = path.size(); + while (end > 1 && path[end - 1] == '/') { + --end; + } + const size_t pos = path.rfind('/', end - 1); + if (pos == std::string::npos) { + return "."; + } + if (pos == 0) { + return std::string("/"); + } + return path.substr(0, pos); +} + +std::string rvs_path_basename(const std::string& path) { + if (path.empty()) { + return path; + } + size_t end = path.size(); + while (end > 0 && path[end - 1] == '/') { + --end; + } + if (end == 0) { + return std::string(); + } + const size_t start = path.rfind('/', end - 1); + if (start == std::string::npos) { + return path.substr(0, end); + } + return path.substr(start + 1, end - start - 1); +} + +#if defined(__linux__) +/** + * @brief Absolute path of the running rvs executable. + * + * Reads /proc/self/exe and resolves symlinks via realpath(). Returns empty + * string on failure. + */ +std::string rvs_linux_resolved_exe_path() { + char linkbuf[4096] = {0}; + const ssize_t n = readlink("/proc/self/exe", linkbuf, sizeof(linkbuf) - 1); + if (n <= 0) { + return std::string(); + } + linkbuf[n] = '\0'; + char resolved[4096] = {0}; + if (realpath(linkbuf, resolved) != nullptr) { + return std::string(resolved); + } + return std::string(linkbuf); +} +#endif + +/** + * @brief RVS install prefix derived from the running process. + * + * On Linux, if the executable is prefix/bin/rvs (or prefix/sbin/rvs), + * returns prefix. Does not consult RVS_PREFIX or other environment variables. + * Returns empty when the prefix cannot be inferred; callers fall back to + * build-time RVS_DATA_ROOT / RVS_LIB_PATH. + */ +std::string rvs_rvs_prefix_from_rvs_process() { +#if defined(__linux__) + const std::string exe = rvs_linux_resolved_exe_path(); + if (exe.empty() || rvs_path_basename(exe) != "rvs") { + return std::string(); + } + const std::string bindir = rvs_path_dirname(exe); + const std::string binname = rvs_path_basename(bindir); + if (binname == "bin" || binname == "sbin") { + return rvs_path_dirname(bindir); + } +#endif + return std::string(); +} + +} // namespace + +/** + * @brief Directory containing RVS shared data (conf files, etc.). + * + * Resolution: install prefix from rvs_rvs_prefix_from_rvs_process() plus + * RVS_RELPATH_DATA_DIR, else compile-time RVS_DATA_ROOT. For configs outside + * this tree, use the -c command-line option. + */ +std::string rvs_get_rvs_data_root_string(void) { + const std::string pfx = rvs_rvs_prefix_from_rvs_process(); + if (!pfx.empty()) { + return pfx + std::string("/") + RVS_RELPATH_DATA_DIR; + } + return std::string(RVS_DATA_ROOT); +} + +/** + * @brief Directory containing RVS module shared libraries (*.so). + * + * Resolution: install prefix from rvs_rvs_prefix_from_rvs_process() plus + * RVS_RELPATH_MODULE_LIB_DIR, else compile-time RVS_LIB_PATH. Used as the + * final fallback in rvsmodule.cpp after relative search paths fail. + */ +std::string rvs_get_rvs_modules_lib_dir_string(void) { + const std::string pfx = rvs_rvs_prefix_from_rvs_process(); + if (!pfx.empty()) { + return pfx + std::string("/") + RVS_RELPATH_MODULE_LIB_DIR; + } + return std::string(RVS_LIB_PATH); +} + +#if defined(__linux__) +/** + * @brief Resolve path to an absolute, symlink-free form. + * + * Wrapper around realpath(3). Used before stat/dlopen so module paths cannot + * be redirected through symlinks. + * + * @param path Input path (may be relative). + * @param out Receives the canonical absolute path on success. + * @return true if resolution succeeded, false otherwise. + */ +bool rvs_canonical_path(const std::string& path, std::string* out) { + if (!out) { + return false; + } + char resolved[4096] = {0}; + if (realpath(path.c_str(), resolved) == nullptr) { + return false; + } + *out = std::string(resolved); + return true; +} +#endif + +/** + * @brief Validate a module .so before dlopen(). + * + * Security gate applied to every module load attempt. On Linux, checks that + * the target is a regular file, is not world-writable, and passes ownership + * and permission rules: group-writable is allowed only when the caller owns + * the file (typical local builds); when euid is 0, group/world writable + * modules are rejected and root ownership is required. Returns the canonical + * path so dlopen uses a stable, resolved location. + * + * @param path Candidate .so path (relative or absolute). + * @param err_msg Optional; receives a short reason on failure. + * @param canonical_path Optional; receives realpath result on success. + * @return true if the file passes all checks, false otherwise. + */ +bool rvs_verify_module_so_for_dlopen(const std::string& path, + std::string* err_msg, + std::string* canonical_path) { +#if defined(__linux__) + std::string canonical; + if (!rvs_canonical_path(path, &canonical)) { + if (err_msg) { + *err_msg = "could not resolve module path"; + } + return false; + } + + struct stat st; + if (stat(canonical.c_str(), &st) != 0) { + if (err_msg) { + *err_msg = "stat failed"; + } + return false; + } + + if (!S_ISREG(st.st_mode)) { + if (err_msg) { + *err_msg = "not a regular file"; + } + return false; + } + + if (st.st_mode & S_IWOTH) { + if (err_msg) { + *err_msg = "module is world-writable"; + } + return false; + } + + const uid_t euid = geteuid(); + if (euid == 0) { + if (st.st_mode & S_IWGRP) { + if (err_msg) { + *err_msg = "module is group-writable"; + } + return false; + } + if (st.st_uid != 0) { + if (err_msg) { + *err_msg = "module is not owned by root"; + } + return false; + } + } else { + if (st.st_uid != euid && st.st_uid != 0) { + if (err_msg) { + *err_msg = "module owner mismatch"; + } + return false; + } + // Allow group-writable modules owned by the caller (typical for local builds). + // st_uid is euid or 0 here; reject root-owned, group-writable system modules. + if ((st.st_mode & S_IWGRP) && st.st_uid == 0) { + if (err_msg) { + *err_msg = "root-owned module is group-writable"; + } + return false; + } + } + + if (canonical_path) { + *canonical_path = canonical; + } + return true; +#else + if (canonical_path) { + *canonical_path = path; + } + return true; +#endif +} + diff --git a/src/rvsactionbase.cpp b/src/rvsactionbase.cpp index bd2c954e7..7a10be448 100644 --- a/src/rvsactionbase.cpp +++ b/src/rvsactionbase.cpp @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -41,7 +41,7 @@ #define FLOATING_POINT_REGEX "^[0-9]*\\.?[0-9]+$" // only thse modules have a target and duration based test approach -static const std::set duration_mods {"gst", "iet", "tst", "pebb", "pbqt", "gm"}; +static const std::set duration_mods {"gst", "iet", "tst", "pebb", "pbqt", "pulse"}; using std::cout; using std::endl; using std::string; diff --git a/src/rvshsa.cpp b/src/rvshsa.cpp index 5c040df41..d13e267de 100644 --- a/src/rvshsa.cpp +++ b/src/rvshsa.cpp @@ -529,6 +529,34 @@ int rvs::hsa::FindAgent(const uint32_t Node) { return -1; } +//! max re-reads of async copy profiling data before giving up +static const int MAX_COPY_TIME_READS = 128; + +/** + * @brief Read async copy profiling timestamps for a completed transfer + * + * The timestamps are not always visible at the moment the completion signal + * is observed - the call then succeeds but yields an all-zero record. Re-read + * until the record is populated. + * + * @param signal signal used for the transfer + * @param pTime [out] profiling timestamps + * @return HSA status of the last read + * + * */ +static hsa_status_t get_async_copy_time(hsa_signal_t signal, + hsa_amd_profiling_async_copy_time_t* pTime) { + hsa_status_t status = HSA_STATUS_SUCCESS; + + for (int i = 0; i < MAX_COPY_TIME_READS; i++) { + status = hsa_amd_profiling_get_async_copy_time(signal, pTime); + if (status != HSA_STATUS_SUCCESS || pTime->end > pTime->start) { + break; + } + } + return status; +} + /** * @brief Fetch time needed to copy data between two memory pools * @@ -546,8 +574,7 @@ double rvs::hsa::GetCopyTime(bool bidirectional, // Obtain time taken for forward copy hsa_amd_profiling_async_copy_time_t async_time_fwd {0}; if (HSA_STATUS_SUCCESS != - (status = - hsa_amd_profiling_get_async_copy_time(signal_fwd, &async_time_fwd))) + (status = get_async_copy_time(signal_fwd, &async_time_fwd))) print_hsa_status(__FILE__, __LINE__, __func__, "hsa_amd_profiling_get_async_copy_time(forward)", status); @@ -559,8 +586,7 @@ double rvs::hsa::GetCopyTime(bool bidirectional, hsa_amd_profiling_async_copy_time_t async_time_rev {0}; if (HSA_STATUS_SUCCESS != - (status = - hsa_amd_profiling_get_async_copy_time(signal_rev, &async_time_rev))) + (status = get_async_copy_time(signal_rev, &async_time_rev))) print_hsa_status(__FILE__, __LINE__, __func__, "hsa_amd_profiling_get_async_copy_time(backward)", status); @@ -744,7 +770,7 @@ int rvs::hsa::Allocate(int SrcAgent, int DstAgent, size_t Size, int rvs::hsa::SendTraffic(uint32_t SrcNode, uint32_t DstNode, size_t Size, bool bidirectional, bool b2b, uint32_t warm_calls, uint32_t hot_calls, - double* Duration) { + double* Duration, uint32_t* NumTimed) { hsa_status_t status; int sts; @@ -781,6 +807,9 @@ int rvs::hsa::SendTraffic(uint32_t SrcNode, uint32_t DstNode, *Duration = 0; + // number of hot calls for which a valid duration could be obtained + uint32_t num_timed = 0; + // Total transfer iterations for (uint32_t i = 0; i < (warm_calls + hot_calls); i++) { @@ -880,8 +909,15 @@ int rvs::hsa::SendTraffic(uint32_t SrcNode, uint32_t DstNode, // Per transfer duration duration = GetCopyTime(bidirectional, signal_fwd, signal_rev)/1000000000; - // Total cumulative duration of all the hot call transfers - *Duration += duration; + // The profiling record is occasionally still unpopulated when the + // completion signal is observed, yielding a zero duration. Such a + // transfer must not contribute to the byte count either, otherwise the + // reported bandwidth is inflated by the untimed bytes. + if (duration > 0) { + // Total cumulative duration of all the hot call transfers + *Duration += duration; + num_timed++; + } } if (!b2b) { @@ -912,6 +948,10 @@ int rvs::hsa::SendTraffic(uint32_t SrcNode, uint32_t DstNode, } } + if (NumTimed) { + *NumTimed = num_timed; + } + RVSHSATRACE_ return 0; diff --git a/src/rvsliblogger.cpp b/src/rvsliblogger.cpp index 383c8d664..fccd32702 100644 --- a/src/rvsliblogger.cpp +++ b/src/rvsliblogger.cpp @@ -23,8 +23,6 @@ * *******************************************************************************/ #include "include/rvsliblogger.h" -#include -#include #include #include @@ -38,6 +36,7 @@ #include #include #include +#include #include "include/rvstrace.h" #include "include/rvslognode.h" @@ -85,17 +84,16 @@ bool isPathedFile(const std::string &fname){ bool doesFolderExist(const std::string &fname){ auto loc = fname.find_last_of('/'); - auto dirName = fname.substr(0,loc); - DIR* dir = opendir(dirName.c_str()); - if (dir == NULL) { - // try creating directory, this doesnt exist. if fails return - std::string command{"mkdir -p "}; - command += dirName; - int ret = system(command.c_str()); - if (ret){ - return false; - } - } + if (loc == std::string::npos) { + return true; + } + auto dirName = fname.substr(0, loc); + std::error_code ec; + std::filesystem::create_directories(dirName, ec); + if (ec) { + cerr << "RVS folder creation error: " << ec.message() << std::endl; + return false; + } std::fstream fs; fs.open(fname, std::ios::out | std::ios::trunc); if (fs.fail()){// unable to create file in dir @@ -140,7 +138,8 @@ bool rvs::logger::append() { } void rvs::logger::set_log_file(const std::string& fname) { - strncpy(log_file, fname.c_str(), sizeof(log_file)); + strncpy(log_file, fname.c_str(), sizeof(log_file) - 1); + log_file[sizeof(log_file) - 1] = '\0'; if (isPathedFile(log_file)){ if (!doesFolderExist(log_file)){ std::cout << "Unable to create log file, check path."; @@ -427,7 +426,11 @@ int rvs::logger::JsonActionStartNodeCreate(const char* Module, const char* Actio void* rvs::logger::JsonNamedListCreate(const char* name,const int LogLevel){ rvs::LogListNode* rec = new rvs::LogListNode(name, LogLevel); return static_cast(rec); +} +void* rvs::logger::JsonNestedListCreate(const char* name, const int LogLevel){ + rvs::LogListNode* rec = new rvs::LogListNode(name, LogLevel, true); + return static_cast(rec); } int rvs::logger::JsonActionEndNodeCreate() { std::string row{RVSINDENT}; diff --git a/src/rvsloglp.cpp b/src/rvsloglp.cpp index a8052c3b0..d94ac18c0 100644 --- a/src/rvsloglp.cpp +++ b/src/rvsloglp.cpp @@ -56,6 +56,7 @@ int rvs::lp::Initialize(const T_MODULE_INIT* pMi) { mi.cbStopping = pMi->cbStopping; mi.cbErr = pMi->cbErr; mi.cbJsonNamedListCreate = pMi->cbJsonNamedListCreate; + mi.cbJsonNestedListCreate = pMi->cbJsonNestedListCreate; return 0; } @@ -342,3 +343,7 @@ void* rvs::lp::JsonNamedListCreate(const char* name, const int LogLevel) { return (*mi.cbJsonNamedListCreate)(name, LogLevel); } +void* rvs::lp::JsonNestedListCreate(const char* name, const int LogLevel) { + return (*mi.cbJsonNestedListCreate)(name, LogLevel); +} + diff --git a/src/rvsloglp_utest.cpp b/src/rvsloglp_utest.cpp index e9aac399f..25bfce7df 100644 --- a/src/rvsloglp_utest.cpp +++ b/src/rvsloglp_utest.cpp @@ -52,7 +52,8 @@ int rvs::lp::Initialize(const T_MODULE_INIT* pMi) { mi.cbAddNode = pMi->cbAddNode; mi.cbStop = pMi->cbStop; mi.cbStopping = pMi->cbStopping; - mi.cbErr = pMi->cbErr; + mi.cbErr = pMi->cbErr; + mi.cbJsonNestedListCreate = pMi->cbJsonNestedListCreate; return 0; } @@ -254,6 +255,10 @@ int rvs::lp::Err(const std::string &Message, const std::string &Module) { return rvs::logger::Err(Message.c_str(), Module.c_str(), nullptr); } +void* rvs::lp::JsonNestedListCreate(const char* name, const int LogLevel) { + return rvs::logger::JsonNestedListCreate(name, LogLevel); +} + /** * @brief Log Error output * diff --git a/src/rvslognodelist.cpp b/src/rvslognodelist.cpp index d7026e291..7a003adeb 100644 --- a/src/rvslognodelist.cpp +++ b/src/rvslognodelist.cpp @@ -37,10 +37,11 @@ using std::string; * @param Parent Pointer to parent node * */ -rvs::LogListNode::LogListNode(const char* Name, int LogLevel, const rvs::LogNodeBase* Parent) +rvs::LogListNode::LogListNode(const char* Name, int LogLevel, bool nested, const rvs::LogNodeBase* Parent) : LogNode(Name, Parent), -Level(LogLevel){ +Level(LogLevel), +IsNested(nested){ Type = eLN::Record; } @@ -84,7 +85,9 @@ void rvs::LogListNode::Add(rvs::LogNodeBase* pChild) { std::string rvs::LogListNode::ToJson(const std::string& Lead) { DTRACE_ string result(RVSENDL); - result += "{"; + if (!IsNested) { + result += "{"; + } result += Lead + "\"" + Name + "\"" + " : ["; int size = Child.size(); @@ -95,6 +98,8 @@ std::string rvs::LogListNode::ToJson(const std::string& Lead) { } } result += RVSENDL + Lead + "]"; - result += "}"; + if (!IsNested) { + result += "}"; + } return result; } diff --git a/testif.so/CMakeLists.txt b/testif.so/CMakeLists.txt index 239912bbf..81cf44173 100644 --- a/testif.so/CMakeLists.txt +++ b/testif.so/CMakeLists.txt @@ -111,7 +111,7 @@ include_directories(./ ../ pci) # Add directories to look for library files to link link_directories(${RVS_LIB_DIR} ${ROCR_LIB_DIR} ${ROCBLAS_LIB_DIR} ${ASAN_LIB_PATH}) ## additional libraries -set (PROJECT_LINK_LIBS libpthread.so libpci.so libm.so) +set (PROJECT_LINK_LIBS libpthread.so libm.so) ## define source files ## set(SOURCES src/rvs_module.cpp src/action.cpp src/worker.cpp) diff --git a/testscripts/pesm.new.sh b/testscripts/pesm.new.sh deleted file mode 100644 index 958898acd..000000000 --- a/testscripts/pesm.new.sh +++ /dev/null @@ -1,28 +0,0 @@ -# ################################################################################ -# # -# # Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. -# # -# # MIT LICENSE: -# # Permission is hereby granted, free of charge, to any person obtaining a copy of -# # this software and associated documentation files (the "Software"), to deal in -# # the Software without restriction, including without limitation the rights to -# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies -# # of the Software, and to permit persons to whom the Software is furnished to do -# # so, subject to the following conditions: -# # -# # The above copyright notice and this permission notice shall be included in all -# # copies or substantial portions of the Software. -# # -# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE -# # SOFTWARE. -# # -# ############################################################################### - -#!/bin/sh -date -echo 'pesm_1';../../../bin/rvs -c ../conf/pesm_1.conf -d 3; date diff --git a/testscripts/rvsqa.new.sh b/testscripts/rvsqa.new.sh index 5e75602ce..f884d7a63 100644 --- a/testscripts/rvsqa.new.sh +++ b/testscripts/rvsqa.new.sh @@ -1,6 +1,6 @@ # ################################################################################ # # -# # Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. +# # Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. # # # # MIT LICENSE: # # Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -36,8 +36,6 @@ echo "===========================pebb=========================" sudo ./pebb.new.sh 2>&1 | tee pebb.txt echo "===========================peqt=========================" sudo ./peqt.new.sh 2>&1 | tee peqt.txt -echo "==========================pesm=========================" -sudo ./pesm.new.sh 2>&1 | tee pesm.txt echo "===========================pbqt=========================" sudo ./pbqt.new.sh 2>&1 | tee pbqt.txt echo "===========================memory=========================" diff --git a/testscripts/smqt.new.sh b/testscripts/smqt.new.sh deleted file mode 100644 index 3d9fa0c08..000000000 --- a/testscripts/smqt.new.sh +++ /dev/null @@ -1,31 +0,0 @@ -# ################################################################################ -# # -# # Copyright (c) 2018-2022 Advanced Micro Devices, Inc. All rights reserved. -# # -# # MIT LICENSE: -# # Permission is hereby granted, free of charge, to any person obtaining a copy of -# # this software and associated documentation files (the "Software"), to deal in -# # the Software without restriction, including without limitation the rights to -# # use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies -# # of the Software, and to permit persons to whom the Software is furnished to do -# # so, subject to the following conditions: -# # -# # The above copyright notice and this permission notice shall be included in all -# # copies or substantial portions of the Software. -# # -# # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -# # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -# # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -# # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -# # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -# # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE -# # SOFTWARE. -# # -# ############################################################################### - -#!/bin/sh -date -echo 'smqt_1';sudo ../../../bin/rvs -c ../conf/smqt_1.conf -d 3; date -echo 'smqt_2';sudo ../../../bin/rvs -c ../conf/smqt_2.conf -d 3; date -echo 'smqt_3';sudo ../../../bin/rvs -c ../conf/smqt_3.conf -d 3; date - diff --git a/tst.so/CMakeLists.txt b/tst.so/CMakeLists.txt index f604f1ba6..ead3e651c 100644 --- a/tst.so/CMakeLists.txt +++ b/tst.so/CMakeLists.txt @@ -158,7 +158,7 @@ include_directories(./ ../ ${AMD_SMI_INC_DIR} ${ROCBLAS_INC_DIR} ${ROCR_INC_DIR} # Add directories to look for library files to link link_directories(${RVS_LIB_DIR} ${ROCR_LIB_DIR} ${ROCBLAS_LIB_DIR} ${AMD_SMI_LIB_DIR} ${ASAN_LIB_PATH}) ## additional libraries -set (PROJECT_LINK_LIBS rvslib libpthread.so libpci.so libm.so) +set (PROJECT_LINK_LIBS rvslib libpthread.so ${LIBPCI_TARGET} libm.so) set(SOURCES src/rvs_module.cpp src/action.cpp src/tst_worker.cpp ) diff --git a/tst.so/include/action.h b/tst.so/include/action.h index 56012b3a7..98e7e2efe 100644 --- a/tst.so/include/action.h +++ b/tst.so/include/action.h @@ -25,13 +25,6 @@ #ifndef TST_SO_INCLUDE_ACTION_H_ #define TST_SO_INCLUDE_ACTION_H_ -#ifdef __cplusplus -extern "C" { -#endif -#include -#ifdef __cplusplus -} -#endif #include #include diff --git a/tst.so/src/action.cpp b/tst.so/src/action.cpp index b52220a5a..dfac0ab67 100644 --- a/tst.so/src/action.cpp +++ b/tst.so/src/action.cpp @@ -34,13 +34,6 @@ #include #include -#ifdef __cplusplus -extern "C" { -#endif -#include -#ifdef __cplusplus -} -#endif #include #define __HIP_PLATFORM_HCC__ diff --git a/tst.so/src/rvs_module.cpp b/tst.so/src/rvs_module.cpp index e2b5c453b..ae173fd93 100644 --- a/tst.so/src/rvs_module.cpp +++ b/tst.so/src/rvs_module.cpp @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2024 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2025 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -72,7 +72,6 @@ extern "C" int rvs_module_init(void* pMi) { } extern "C" int rvs_module_terminate(void) { - cleanup_logs(); amdsmi_shut_down(); return 0; } diff --git a/tst.so/src/tst_worker.cpp b/tst.so/src/tst_worker.cpp index b4e2c2456..eea96fa24 100644 --- a/tst.so/src/tst_worker.cpp +++ b/tst.so/src/tst_worker.cpp @@ -1,6 +1,6 @@ /******************************************************************************** * - * Copyright (c) 2018-2025 Advanced Micro Devices, Inc. All rights reserved. + * Copyright (c) 2018-2026 Advanced Micro Devices, Inc. All rights reserved. * * MIT LICENSE: * Permission is hereby granted, free of charge, to any person obtaining a copy of @@ -154,7 +154,7 @@ void TSTWorker::blasThread(int gpuIdx, uint64_t matrix_size, std::string tst_ops //Hit the GPU with load to increase temperature while ( (duration < run_duration_ms) && (endtest == false) ){ //call the gemm blas - gpu_blas->run_blas_gemm(true); + gpu_blas->run_blas_gemm(1); /* Set callback to be called upon completion of blas gemm operations */ gpu_blas->set_callback(blas_callback, (void *)this);