diff --git a/docker/spark/Dockerfile.spark4 b/docker/spark/Dockerfile.spark4 new file mode 100644 index 00000000..4c825ab7 --- /dev/null +++ b/docker/spark/Dockerfile.spark4 @@ -0,0 +1,33 @@ +# check=skip=InvalidDefaultArgInFrom +# Copyright 2026 Iguazio +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# +# Spark 4 counterpart of ./Dockerfile. +# Spark 4.2.0, Scala 2.13, Temurin JDK 25, Hadoop 3.5.0 as bundled by the +# pinned upstream image. + +# Required build argument; the Makefile supplies the pinned base reference. +ARG SPARK4_BASE_IMAGE +FROM ${SPARK4_BASE_IMAGE} + +USER root + +# Shared with the Spark 4 CUDA image. +COPY scripts/ce-customize-spark4.sh /tmp/ce-customize-spark4.sh +COPY scripts/jars-4.2.0.txt /tmp/jars-4.2.0.txt +RUN chmod +x /tmp/ce-customize-spark4.sh && \ + /tmp/ce-customize-spark4.sh /tmp/jars-4.2.0.txt && \ + rm -f /tmp/ce-customize-spark4.sh /tmp/jars-4.2.0.txt + +USER spark diff --git a/docker/spark/Dockerfile.spark4.cuda b/docker/spark/Dockerfile.spark4.cuda new file mode 100644 index 00000000..57767845 --- /dev/null +++ b/docker/spark/Dockerfile.spark4.cuda @@ -0,0 +1,87 @@ +# check=skip=InvalidDefaultArgInFrom +# Copyright 2026 Iguazio +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# +# CUDA counterpart of ./Dockerfile.spark4. The Spark distribution, the JDK, +# and the entrypoint are copied from the same digest-pinned base image the +# regular Spark 4 image is built from, so the two variants cannot drift and +# the build does not download Spark or the JDK a second time. + +# Required build arguments; the Makefile supplies the pinned base references. +ARG SPARK4_BASE_IMAGE +ARG SPARK4_CUDA_BASE_IMAGE + +FROM ${SPARK4_BASE_IMAGE} AS spark4 + +FROM ${SPARK4_CUDA_BASE_IMAGE} + +# CUDA_VERSION and NV_CUDNN_VERSION are the CUDA base image's own environment. +# Do not declare an ARG of either name here: it would shadow the inherited +# value and let the labels describe a base this image was not built on. +LABEL com.iguazio.cuda-version="${CUDA_VERSION}" \ + com.iguazio.cudnn-version="${NV_CUDNN_VERSION}" + +ARG spark_uid=185 + +RUN groupadd --system --gid=${spark_uid} spark && \ + useradd --system --uid=${spark_uid} --gid=spark -d /nonexistent spark + +# Packages the upstream Spark image provides and the CUDA base does not. +# `locales` is one of them: the CUDA base generates only C and C.utf8, so +# en_US.UTF-8 has to be generated here for the locale ENV below to be valid. +RUN set -ex; \ + export DEBIAN_FRONTEND=noninteractive; \ + apt-get update; \ + apt-get install -y --no-install-recommends \ + gnupg2 bash tini libc6 libpam-modules \ + krb5-user libnss3 procps net-tools gosu libnss-wrapper libjemalloc2 \ + locales; \ + locale-gen en_US.UTF-8; \ + echo "auth required pam_wheel.so use_uid" >> /etc/pam.d/su; \ + rm -rf /var/lib/apt/lists/* + +# COPY carries no image environment, so the runtime settings from the Spark +# base are restated here. SPARK_TGZ_URL, SPARK_TGZ_ASC_URL, and GPG_KEY are +# build-provenance metadata and are intentionally omitted. +ENV LANG=en_US.UTF-8 \ + LANGUAGE=en_US:en \ + LC_ALL=en_US.UTF-8 + +COPY --from=spark4 /opt/java/openjdk /opt/java/openjdk +COPY --from=spark4 /opt/spark /opt/spark +COPY --from=spark4 /opt/decom.sh /opt/decom.sh +COPY --from=spark4 /opt/entrypoint.sh /opt/entrypoint.sh + +# COPY creates the destination directory itself as root, while its contents +# keep the source image's ownership. Restore the Spark base's owner on the +# directory so the two variants match exactly. +RUN chown spark:spark /opt/spark + +ENV JAVA_HOME=/opt/java/openjdk \ + JAVA_VERSION=jdk-25.0.4+7 \ + SPARK_HOME=/opt/spark +ENV PATH="${JAVA_HOME}/bin:${PATH}" + +WORKDIR /opt/spark/work-dir + +# Shared with the regular Spark 4 image. +COPY scripts/ce-customize-spark4.sh /tmp/ce-customize-spark4.sh +COPY scripts/jars-4.2.0.txt /tmp/jars-4.2.0.txt +RUN chmod +x /tmp/ce-customize-spark4.sh && \ + /tmp/ce-customize-spark4.sh /tmp/jars-4.2.0.txt && \ + rm -f /tmp/ce-customize-spark4.sh /tmp/jars-4.2.0.txt + +USER spark + +ENTRYPOINT [ "/opt/entrypoint.sh" ] diff --git a/docker/spark/Makefile b/docker/spark/Makefile index 3c515742..994616af 100644 --- a/docker/spark/Makefile +++ b/docker/spark/Makefile @@ -27,3 +27,47 @@ validate-cuda: .PHONY: validate-all validate-all: validate validate-cuda + +# Spark 4 variables and targets. +REGISTRY ?= gcr.io/iguazio +# Spark uses a multi-platform index; CUDA uses its linux/amd64 child manifest +# because the Spark 4 image pair is intentionally amd64-only. +SPARK4_BASE_IMAGE ?= spark@sha256:66e39dccde81909c23e5c56f4b465db6de2bdc38cf569aacc9e2380ee5005440 +SPARK4_CUDA_BASE_IMAGE ?= nvidia/cuda@sha256:61f6c08f2b59036cb935e56d1e31a6b64e3ae2c7ddb86d33fa0b044c7917b719 +MLRUN_CE_SPARK4_IMAGE_TAG ?= $(REGISTRY)/spark-app:4.2.0-scala2.13-java25-ubuntu-1 +MLRUN_CE_SPARK4_CUDA_IMAGE_TAG ?= $(REGISTRY)/spark-app-cuda:4.2.0-scala2.13-java25-ubuntu-1 + +.PHONY: build-spark4 +build-spark4: + docker build --platform=linux/amd64 \ + --build-arg SPARK4_BASE_IMAGE="$(SPARK4_BASE_IMAGE)" \ + -t "$(MLRUN_CE_SPARK4_IMAGE_TAG)" -f Dockerfile.spark4 . + +.PHONY: build-spark4-cuda +build-spark4-cuda: + docker build --platform=linux/amd64 \ + --build-arg SPARK4_BASE_IMAGE="$(SPARK4_BASE_IMAGE)" \ + --build-arg SPARK4_CUDA_BASE_IMAGE="$(SPARK4_CUDA_BASE_IMAGE)" \ + -t "$(MLRUN_CE_SPARK4_CUDA_IMAGE_TAG)" -f Dockerfile.spark4.cuda . + +.PHONY: build-spark4-all +build-spark4-all: build-spark4 build-spark4-cuda + +.PHONY: validate-spark4 +validate-spark4: + bash scripts/validate-spark4.sh "$(MLRUN_CE_SPARK4_IMAGE_TAG)" + +.PHONY: validate-spark4-cuda +validate-spark4-cuda: + bash scripts/validate-spark4.sh "$(MLRUN_CE_SPARK4_CUDA_IMAGE_TAG)" --cuda + +.PHONY: validate-spark4-all +validate-spark4-all: validate-spark4 validate-spark4-cuda jar-parity-spark4 env-parity-spark4 + +.PHONY: jar-parity-spark4 +jar-parity-spark4: + bash scripts/jar-parity.sh "$(MLRUN_CE_SPARK4_IMAGE_TAG)" "$(MLRUN_CE_SPARK4_CUDA_IMAGE_TAG)" + +.PHONY: env-parity-spark4 +env-parity-spark4: + bash scripts/env-parity.sh "$(MLRUN_CE_SPARK4_IMAGE_TAG)" "$(MLRUN_CE_SPARK4_CUDA_IMAGE_TAG)" diff --git a/docker/spark/README.md b/docker/spark/README.md index 3bc3d8be..434d11c8 100644 --- a/docker/spark/README.md +++ b/docker/spark/README.md @@ -49,3 +49,104 @@ docker inspect --format '{{index .RepoDigests 0}}' \ Do not republish the existing regular image. Record the CUDA image digest and the CE source commit. + +## Spark 4 + +The Spark 4 images target `linux/amd64` and use Spark 4.2.0, Scala 2.13, +Hadoop 3.5.0, Temurin 25.0.4+7, and Python 3.11: + +- `spark-app:4.2.0-scala2.13-java25-ubuntu-1` uses the digest-pinned Spark + base configured by `SPARK4_BASE_IMAGE` in the Makefile. +- `spark-app-cuda:4.2.0-scala2.13-java25-ubuntu-1` copies Spark and Java from + that base into the digest-pinned `SPARK4_CUDA_BASE_IMAGE` (CUDA 12.8.1, + cuDNN 9.8.0.87-1). + +Both images install the connector artifacts listed in +`scripts/jars-4.2.0.txt`: Hadoop 3.5.0 connectors for S3A, ABFS, and GCS, +their required AWS and Azure dependencies, and the Scala 2.13 BigQuery +connector. These restore the connector family shipped by the published Spark +3 image so Spark 4 can serve as a like-for-like platform image. Local +validation checks packaging, class resolution, duplicate versions, and +CPU/CUDA parity. Authenticated provider testing and BigQuery compatibility +with Spark 4.2 are deferred to the corresponding activation work. + +`hadoop-gcp-3.5.0` provides +`org.apache.hadoop.fs.gs.GoogleHadoopFileSystem`, replacing the former +`com.google.cloud.hadoop.fs.gcs.GoogleHadoopFileSystem` class. It does not +provide a replacement `AbstractFileSystem` implementation. Consumers using +the old `fs.gs.impl` or `fs.AbstractFileSystem.gs.impl` configuration must +update or remove it. + +`hadoop-aws-3.5.0` uses AWS SDK v2. Hadoop remaps several common SDK v1 +credential-provider names, but not +`com.amazonaws.auth.DefaultAWSCredentialsProviderChain`, arbitrary +`com.amazonaws.*` providers, or custom SDK v1 implementations. Consumers +using those providers must migrate their configuration. + +MLRun selects the CPU repository and tag through `MLRUN_SPARK_APP_IMAGE` and +`MLRUN_SPARK_APP_IMAGE_TAG`, and derives the CUDA repository by appending +`-cuda`. mlefi resolves both repositories and extracts the Spark version from +the tag with `^(.+?)-scala.*$`; for example, +`4.2.0-scala2.13-java25-ubuntu-1` yields `4.2.0`. These recipes do not change +the configured defaults. + +### Build + +```bash +make build-spark4 +make build-spark4-cuda +make build-spark4-all +``` + +`REGISTRY` defaults to `gcr.io/iguazio`. The Makefile is the source of truth +for both pinned base-image digests. + +### Validate + +```bash +make validate-spark4-all +``` + +The validators check Spark, Scala, Java, Hadoop, Python, the connector +inventory, image metadata, and CUDA metadata. The aggregate target also checks +CPU/CUDA environment parity and compares every `$SPARK_HOME/jars` entry by +SHA-256. It does not authenticate to AWS, Azure, or GCP services. + +### Publish (manual, JFrog) + +Publication is manual. First check that neither +`spark-app:4.2.0-scala2.13-java25-ubuntu-1` nor +`spark-app-cuda:4.2.0-scala2.13-java25-ubuntu-1` already exists in +`mckinsey-ig4-next-gen-docker-local.jfrog.io`. Never overwrite an existing +immutable tag. + +```bash +make REGISTRY=mckinsey-ig4-next-gen-docker-local.jfrog.io build-spark4-all +make REGISTRY=mckinsey-ig4-next-gen-docker-local.jfrog.io validate-spark4-all + +docker login mckinsey-ig4-next-gen-docker-local.jfrog.io +docker push mckinsey-ig4-next-gen-docker-local.jfrog.io/spark-app:4.2.0-scala2.13-java25-ubuntu-1 +docker push mckinsey-ig4-next-gen-docker-local.jfrog.io/spark-app-cuda:4.2.0-scala2.13-java25-ubuntu-1 + +docker inspect --format '{{index .RepoDigests 0}}' \ + mckinsey-ig4-next-gen-docker-local.jfrog.io/spark-app:4.2.0-scala2.13-java25-ubuntu-1 +docker inspect --format '{{index .RepoDigests 0}}' \ + mckinsey-ig4-next-gen-docker-local.jfrog.io/spark-app-cuda:4.2.0-scala2.13-java25-ubuntu-1 +``` + +Record both repository digests, then pull each image back by digest and +rerun validation against the digest references. A local image ID is not a +published digest. + +### Jira evidence (ML-13080) + +Attach to ML-13080: + +- immutable tags and repository digests; +- pull-by-digest and image-inspection output; +- `spark-submit --version` and `java -version` output; +- CPU and CUDA JAR SHA-256 inventories and the parity result; +- the pinned base-image references from the Makefile; +- the source commit; +- the connector inventory and deferred provider-validation status documented + above. diff --git a/docker/spark/scripts/ce-customize-spark4.sh b/docker/spark/scripts/ce-customize-spark4.sh new file mode 100644 index 00000000..f3a2e27e --- /dev/null +++ b/docker/spark/scripts/ce-customize-spark4.sh @@ -0,0 +1,50 @@ +#!/bin/bash +# Copyright 2026 Iguazio +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# +# Spark 4 counterpart of ce-customize.sh. Installs Python 3.11, connector JARs +# from the supplied manifest, and a writable home directory for the spark user. +set -ex +export DEBIAN_FRONTEND=noninteractive + +JAR_MANIFEST="${1:?usage: ce-customize-spark4.sh }" + +apt-get update +apt-get install -y --no-install-recommends software-properties-common curl ca-certificates gnupg +add-apt-repository -y ppa:deadsnakes/ppa +apt-get update +apt-get install -y python3.11 python3.11-distutils git +rm -rf /var/lib/apt/lists/* + +curl https://bootstrap.pypa.io/get-pip.py -o /tmp/get-pip.py +python3.11 /tmp/get-pip.py +rm -f /tmp/get-pip.py + +while read -r line || [ -n "$line" ]; do + url="${line%%#*}" + url="$(echo "$url" | tr -d '[:space:]')" + [ -z "$url" ] && continue + case "$url" in + https://*) ;; + *) echo "bad manifest line: $line" >&2; exit 1 ;; + esac + curl -fsSL -o "/opt/spark/jars/$(basename "$url")" "$url" +done < "$JAR_MANIFEST" + +update-alternatives --install /usr/bin/python3 python3 /usr/bin/python3.11 1 +ln -sf /usr/bin/python3 /usr/bin/python + +usermod -d /home/spark spark +mkdir -p /home/spark +chown spark:spark /home/spark diff --git a/docker/spark/scripts/env-parity.sh b/docker/spark/scripts/env-parity.sh new file mode 100644 index 00000000..3355fa63 --- /dev/null +++ b/docker/spark/scripts/env-parity.sh @@ -0,0 +1,110 @@ +#!/bin/bash + +# Copyright 2026 Iguazio +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# +# Compares CPU and CUDA image environments. CUDA runtime variables and the +# intentional omission of Spark build-provenance variables are allowed. + +set -euo pipefail + +CPU_IMAGE="${1:?usage: env-parity.sh }" +CUDA_IMAGE="${2:?usage: env-parity.sh }" + +image_env() { + docker inspect "$1" --format '{{range .Config.Env}}{{println .}}{{end}}' +} + +lookup() { + local env_text="$1" + local key="$2" + local line + + while IFS= read -r line; do + if [[ "${line%%=*}" == "$key" ]]; then + printf '%s\n' "${line#*=}" + return 0 + fi + done <<<"$env_text" + return 1 +} + +cpu_env="$(image_env "$CPU_IMAGE")" +cuda_env="$(image_env "$CUDA_IMAGE")" + +echo "==> comparing shared image environment" +while IFS= read -r line; do + [[ -n "$line" ]] || continue + key="${line%%=*}" + cpu_value="${line#*=}" + + case "$key" in + SPARK_TGZ_URL|SPARK_TGZ_ASC_URL|GPG_KEY) + if lookup "$cuda_env" "$key" >/dev/null; then + echo "FAIL: CUDA image unexpectedly contains $key" + exit 1 + fi + ;; + PATH) + cuda_value="$(lookup "$cuda_env" "$key")" || { + echo "FAIL: CUDA image is missing $key" + exit 1 + } + [[ ":$cuda_value:" == *":/usr/local/cuda/bin:"* ]] || { + echo "FAIL: CUDA PATH is missing /usr/local/cuda/bin" + exit 1 + } + normalized_cuda_path="${cuda_value/:\/usr\/local\/cuda\/bin/}" + [[ "$normalized_cuda_path" == "$cpu_value" ]] || { + echo "FAIL: PATH differs beyond the expected CUDA addition" + echo "CPU: $cpu_value" + echo "CUDA: $cuda_value" + exit 1 + } + ;; + *) + cuda_value="$(lookup "$cuda_env" "$key")" || { + echo "FAIL: CUDA image is missing $key" + exit 1 + } + [[ "$cuda_value" == "$cpu_value" ]] || { + echo "FAIL: $key differs" + echo "CPU: $cpu_value" + echo "CUDA: $cuda_value" + exit 1 + } + ;; + esac +done <<<"$cpu_env" + +echo "==> checking CUDA-only environment allowlist" +while IFS= read -r line; do + [[ -n "$line" ]] || continue + key="${line%%=*}" + + if lookup "$cpu_env" "$key" >/dev/null; then + continue + fi + + case "$key" in + NV_*|CUDA_*|NVIDIA_*|NVARCH|NCCL_VERSION|LD_LIBRARY_PATH|LIBRARY_PATH) + ;; + *) + echo "FAIL: unexpected CUDA-only environment variable: $key" + exit 1 + ;; + esac +done <<<"$cuda_env" + +echo "==> CPU/CUDA environment parity passed" diff --git a/docker/spark/scripts/jar-parity.sh b/docker/spark/scripts/jar-parity.sh new file mode 100644 index 00000000..62d6d128 --- /dev/null +++ b/docker/spark/scripts/jar-parity.sh @@ -0,0 +1,46 @@ +#!/bin/bash + +# Copyright 2026 Iguazio +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# +# Asserts that two images carry the same $SPARK_HOME/jars by content, not just +# by filename. + +set -euo pipefail + +IMAGE_A="${1:?usage: jar-parity.sh }" +IMAGE_B="${2:?usage: jar-parity.sh }" + +checksums() { + # LC_ALL=C: the two images may differ in locale, and collation order would + # otherwise diff even when the contents match. + docker run --rm --platform linux/amd64 \ + --entrypoint bash "$1" -c \ + 'cd "$SPARK_HOME/jars" && LC_ALL=C sha256sum *.jar | LC_ALL=C sort' +} + +echo "==> comparing \$SPARK_HOME/jars: $IMAGE_A vs $IMAGE_B" +a="$(checksums "$IMAGE_A")" +b="$(checksums "$IMAGE_B")" + +count="$(wc -l <<<"$a" | tr -d '[:space:]')" +[[ "$count" -gt 0 ]] || { echo "FAIL: no JARs found in $IMAGE_A"; exit 1; } + +if [[ "$a" != "$b" ]]; then + echo "FAIL: JAR contents differ:" + diff <(echo "$a") <(echo "$b") || true + exit 1 +fi + +echo "==> $count JARs identical by SHA-256" diff --git a/docker/spark/scripts/jars-4.2.0.txt b/docker/spark/scripts/jars-4.2.0.txt new file mode 100644 index 00000000..1ed9b496 --- /dev/null +++ b/docker/spark/scripts/jars-4.2.0.txt @@ -0,0 +1,8 @@ +https://repo1.maven.org/maven2/org/apache/hadoop/hadoop-aws/3.5.0/hadoop-aws-3.5.0.jar +https://repo1.maven.org/maven2/software/amazon/s3/analyticsaccelerator/analyticsaccelerator-s3/1.3.1/analyticsaccelerator-s3-1.3.1.jar +https://repo1.maven.org/maven2/org/apache/hadoop/hadoop-azure/3.5.0/hadoop-azure-3.5.0.jar +https://repo1.maven.org/maven2/software/amazon/awssdk/bundle/2.35.4/bundle-2.35.4.jar +https://repo1.maven.org/maven2/org/wildfly/openssl/wildfly-openssl/2.2.5.Final/wildfly-openssl-2.2.5.Final.jar +https://repo1.maven.org/maven2/com/microsoft/azure/azure-storage/7.0.1/azure-storage-7.0.1.jar +https://repo1.maven.org/maven2/org/apache/hadoop/hadoop-gcp/3.5.0/hadoop-gcp-3.5.0.jar +https://repo1.maven.org/maven2/com/google/cloud/spark/spark-bigquery-with-dependencies_2.13/0.45.0/spark-bigquery-with-dependencies_2.13-0.45.0.jar diff --git a/docker/spark/scripts/validate-spark4.sh b/docker/spark/scripts/validate-spark4.sh new file mode 100644 index 00000000..863832aa --- /dev/null +++ b/docker/spark/scripts/validate-spark4.sh @@ -0,0 +1,144 @@ +#!/bin/bash + +# Copyright 2026 Iguazio +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# +# Spark 4 counterpart of validate.sh. Local image checks that do not require +# a GPU. + +set -euo pipefail + +IMAGE="${1:?usage: validate-spark4.sh [--cuda]}" +CUDA_MODE="${2:-}" + +PINNED_CUDA_VERSION="12.8.1" +PINNED_CUDNN_VERSION="9.8.0.87-1" +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +JAR_MANIFEST="$SCRIPT_DIR/jars-4.2.0.txt" + +EXPECTED_BASE_JARS=( + hadoop-client-api-3.5.0.jar + hadoop-client-runtime-3.5.0.jar +) + +[[ -f "$JAR_MANIFEST" ]] || { echo "missing JAR manifest: $JAR_MANIFEST" >&2; exit 1; } +EXPECTED_CONNECTOR_JARS=() +while read -r line || [ -n "$line" ]; do + url="${line%%#*}" + url="$(echo "$url" | tr -d '[:space:]')" + [ -z "$url" ] && continue + EXPECTED_CONNECTOR_JARS+=("$(basename "$url")") +done < "$JAR_MANIFEST" +[[ ${#EXPECTED_CONNECTOR_JARS[@]} -gt 0 ]] || { echo "no connector JARs listed in $JAR_MANIFEST" >&2; exit 1; } + +run() { + docker run --rm --platform linux/amd64 --entrypoint bash "$IMAGE" -c "$1" +} + +echo "==> [$IMAGE] linux/amd64 architecture" +arch="$(docker inspect "$IMAGE" --format '{{.Os}}/{{.Architecture}}')" +[[ "$arch" == "linux/amd64" ]] || { echo "FAIL: unexpected architecture: '$arch'"; exit 1; } +uname_m="$(run 'uname -m')" +[[ "$uname_m" == "x86_64" ]] || { echo "FAIL: unexpected uname -m: '$uname_m'"; exit 1; } + +echo "==> [$IMAGE] preserves the Spark entrypoint" +entrypoint="$(docker inspect "$IMAGE" --format '{{json .Config.Entrypoint}}')" +[[ "$entrypoint" == '["/opt/entrypoint.sh"]' ]] || { echo "FAIL: unexpected entrypoint: '$entrypoint'"; exit 1; } + +echo "==> [$IMAGE] runs as the spark user" +whoami="$(run 'whoami')" +[[ "$whoami" == "spark" ]] || { echo "FAIL: expected spark user, got '$whoami'"; exit 1; } + +echo "==> [$IMAGE] working spark-submit / Spark 4.2.0 / Scala 2.13" +spark_version="$(run '$SPARK_HOME/bin/spark-submit --version 2>&1')" +grep -q 'version 4.2.0' <<<"$spark_version" || { echo "FAIL: Spark is not version 4.2.0:"; echo "$spark_version"; exit 1; } +grep -q 'Scala version 2.13' <<<"$spark_version" || { echo "FAIL: Scala is not version 2.13:"; echo "$spark_version"; exit 1; } + +echo "==> [$IMAGE] Temurin 25.0.4+7" +java_version="$(run 'java -version 2>&1')" +grep -Fq 'Temurin-25.0.4+7' <<<"$java_version" || { echo "FAIL: Java is not Temurin 25.0.4+7:"; echo "$java_version"; exit 1; } + +echo "==> [$IMAGE] Hadoop 3.5.0" +for jar in "${EXPECTED_BASE_JARS[@]}"; do + run "test -f \$SPARK_HOME/jars/$jar" || { echo "FAIL: missing jar $jar"; exit 1; } +done + +echo "==> [$IMAGE] ${#EXPECTED_CONNECTOR_JARS[@]} connector JARs" +for jar in "${EXPECTED_CONNECTOR_JARS[@]}"; do + run "test -f \$SPARK_HOME/jars/$jar" || { echo "FAIL: missing connector jar $jar"; exit 1; } +done + +echo "==> [$IMAGE] no duplicate connector versions" +CONNECTOR_JAR_PREFIXES='^(hadoop-aws|hadoop-azure|hadoop-common|hadoop-gcp|aws-java-sdk-bundle|bundle|analyticsaccelerator-s3|wildfly-openssl|azure-storage|gcs-connector|jetty-util|jetty-util-ajax|spark-bigquery-with-dependencies(_[0-9]+\.[0-9]+)?)-' +duplicates="$(run 'ls "$SPARK_HOME/jars"' | grep -E "$CONNECTOR_JAR_PREFIXES" | sed -E 's/(_[0-9]+\.[0-9]+)?-[0-9][A-Za-z0-9._-]*\.jar$//' | sort | uniq -d)" +[[ -z "$duplicates" ]] || { echo "FAIL: duplicate connector artifacts:"; echo "$duplicates"; exit 1; } + +echo "==> [$IMAGE] connector classes resolve" +for class in \ + org.apache.hadoop.fs.s3a.S3AFileSystem \ + org.apache.hadoop.fs.azurebfs.oauth2.WorkloadIdentityTokenProvider \ + org.apache.hadoop.fs.gs.GoogleHadoopFileSystem; do + run "javap -classpath \"\$SPARK_HOME/jars/*\" '$class' >/dev/null" \ + || { echo "FAIL: connector class does not resolve: $class"; exit 1; } +done + +echo "==> [$IMAGE] Python 3.11" +python_version="$(run 'python3 --version 2>&1')" +grep -q 'Python 3.11' <<<"$python_version" || { echo "FAIL: Python is not version 3.11: $python_version"; exit 1; } + +echo "==> [$IMAGE] spark home directory is writable" +passwd_entry="$(run 'getent passwd spark')" +grep -q ':/home/spark:' <<<"$passwd_entry" || { echo "FAIL: spark home is not /home/spark: '$passwd_entry'"; exit 1; } +run 'test -w /home/spark' || { echo "FAIL: /home/spark is not writable by spark"; exit 1; } + +echo "==> [$IMAGE] \$SPARK_HOME ownership" +spark_home_owner="$(run 'stat -c %U:%G $SPARK_HOME')" +[[ "$spark_home_owner" == "spark:spark" ]] || { echo "FAIL: unexpected \$SPARK_HOME ownership: '$spark_home_owner'"; exit 1; } + +echo "==> [$IMAGE] UTF-8 locale" +charmap="$(run 'locale charmap')" +[[ "$charmap" == "UTF-8" ]] || { echo "FAIL: expected UTF-8, got '$charmap'"; exit 1; } + +if [[ "$CUDA_MODE" == "--cuda" ]]; then + echo "==> [$IMAGE] CUDA_VERSION=$PINNED_CUDA_VERSION" + cuda_env="$(docker inspect "$IMAGE" --format '{{range .Config.Env}}{{println .}}{{end}}' | grep '^CUDA_VERSION=' || true)" + [[ "$cuda_env" == "CUDA_VERSION=$PINNED_CUDA_VERSION" ]] || { echo "FAIL: unexpected CUDA_VERSION: '$cuda_env'"; exit 1; } + + echo "==> [$IMAGE] NV_CUDNN_VERSION=$PINNED_CUDNN_VERSION" + cudnn_env="$(docker inspect "$IMAGE" --format '{{range .Config.Env}}{{println .}}{{end}}' | grep '^NV_CUDNN_VERSION=' || true)" + [[ "$cudnn_env" == "NV_CUDNN_VERSION=$PINNED_CUDNN_VERSION" ]] || { echo "FAIL: unexpected NV_CUDNN_VERSION: '$cudnn_env'"; exit 1; } + + echo "==> [$IMAGE] com.iguazio.cuda-version / com.iguazio.cudnn-version labels" + cuda_label="$(docker inspect "$IMAGE" --format '{{index .Config.Labels "com.iguazio.cuda-version"}}')" + [[ "$cuda_label" == "$PINNED_CUDA_VERSION" ]] || { echo "FAIL: unexpected com.iguazio.cuda-version label: '$cuda_label'"; exit 1; } + cudnn_label="$(docker inspect "$IMAGE" --format '{{index .Config.Labels "com.iguazio.cudnn-version"}}')" + [[ "$cudnn_label" == "$PINNED_CUDNN_VERSION" ]] || { echo "FAIL: unexpected com.iguazio.cudnn-version label: '$cudnn_label'"; exit 1; } + + echo "==> [$IMAGE] nvcc release 12.8" + nvcc_version="$(run 'nvcc --version 2>&1')" + grep -Fq 'release 12.8' <<<"$nvcc_version" || { echo "FAIL: CUDA toolkit is not 12.8: $nvcc_version"; exit 1; } + + echo "==> [$IMAGE] cuDNN 9 libraries present" + run 'ldconfig -p | grep -q libcudnn.so.9' || { echo "FAIL: libcudnn.so.9 not found"; exit 1; } + + echo "==> [$IMAGE] NVIDIA_VISIBLE_DEVICES=all" + visible_devices="$(docker inspect "$IMAGE" --format '{{range .Config.Env}}{{println .}}{{end}}' | grep '^NVIDIA_VISIBLE_DEVICES=' || true)" + [[ "$visible_devices" == "NVIDIA_VISIBLE_DEVICES=all" ]] || { echo "FAIL: unexpected NVIDIA_VISIBLE_DEVICES: '$visible_devices'"; exit 1; } + + echo "==> [$IMAGE] NVIDIA_DRIVER_CAPABILITIES=compute,utility" + caps="$(docker inspect "$IMAGE" --format '{{range .Config.Env}}{{println .}}{{end}}' | grep '^NVIDIA_DRIVER_CAPABILITIES=' || true)" + [[ "$caps" == "NVIDIA_DRIVER_CAPABILITIES=compute,utility" ]] || { echo "FAIL: unexpected NVIDIA_DRIVER_CAPABILITIES: '$caps'"; exit 1; } +fi + +echo "==> [$IMAGE] all checks passed"