From 6691f55f4a4d67bb7b73cbe35eed6b6a8fbcaabf Mon Sep 17 00:00:00 2001 From: Gal Topper Date: Mon, 21 Sep 2026 16:24:46 +0700 Subject: [PATCH 1/5] [Feature] Add Spark 4 platform images Adds separately maintained Spark 4.2.0 CPU and CUDA application images for ML-13080 without changing the existing Spark 3 recipes or configured defaults. Both variants use pinned upstream images, Spark 4.2.0, Scala 2.13, Hadoop 3.5.0, Temurin 25.0.4+7, and Python 3.11, with local validation for image metadata and CPU/CUDA JAR parity. --- - Added `docker/spark/Dockerfile.spark4` for the CPU image using a digest-pinned upstream Spark base. - Added `docker/spark/Dockerfile.spark4.cuda` for the CUDA image using digest-pinned Spark and CUDA bases without downloading Spark or Java again. - Added `docker/spark/scripts/ce-customize-spark4.sh` for Python 3.11 and a writable `/home/spark`. - Added `docker/spark/scripts/validate-spark4.sh` for Spark, Scala, Java, Hadoop, Python, architecture, locale, ownership, entrypoint, CUDA, cuDNN, and NVIDIA metadata validation. - Added `docker/spark/scripts/jar-parity.sh` to compare every CPU and CUDA Spark JAR by SHA-256. - Added independent Spark 4 build and validation targets to `docker/spark/Makefile`. - Documented Spark 4 build, validation, connector limitations, selection, manual publication, and evidence capture in `docker/spark/README.md`. --- - [ ] I have tested the changes in this PR - [ ] I confirmed whether my changes require a change in documentation and if so, I created another PR in MLRun for the relevant documentation. - [ ] I confirmed whether my changes require a changes in QA tests, for example: credentials changes, resources naming change and if so, I updated the relevant Jira ticket for QA. - [ ] I increased the Chart version in `charts/mlrun-ce/Chart.yaml`. - [ ] I confirmed that the installation works both on a local Docker Desktop environment and on a real cluster when using the required [prerequisites](https://docs.mlrun.org/en/stable/install-mlrun-ce/kubernetes-install.html#prerequisites). - [ ] If installation issues were found, I updated the relevant Jira ticket with the issue and steps to reproduce, or updated the prerequisites documentation if the issue is related to missing or outdated prerequisites. - [x] If needed, update https://github.com/mlrun/ce/blob/development/charts/mlrun-ce/README.md with the relevant installation instructions and version Matrix. - [x] If needed, update the following values files for multi namespace support: - [x] [Admin values](https://github.com/mlrun/ce/blob/development/charts/mlrun-ce/admin_installation_values.yaml) - [x] [User values Node Port](https://github.com/mlrun/ce/blob/development/charts/mlrun-ce/non_admin_installation_values.yaml) - [x] [User values ClusterIP](https://github.com/mlrun/ce/blob/development/charts/mlrun-ce/non_admin_cluster_ip_installation_values.yaml) --- - Built both `linux/amd64` images with `make build-spark4-all`. - Ran `make validate-spark4-all`; both images passed Spark, Scala, Java, Hadoop, Python, architecture, entrypoint, user, home-directory, ownership, locale, and image-metadata checks. - Validated CUDA 12.8.1, cuDNN 9.8.0.87-1, `nvcc`, cuDNN libraries, labels, and NVIDIA runtime environment metadata. - Ran SparkPi successfully in both images. - Ran `make jar-parity-spark4`; all 276 JARs matched by SHA-256. - Confirmed the protected Spark 3 files have no diff from `development` and the existing Spark 3 Make targets are unchanged. --- - Ticket link: https://ecliptos.atlassian.net/browse/ML-13080 - External links: https://hub.docker.com/_/spark and https://hub.docker.com/r/nvidia/cuda - Design docs links (Optional): `docker/spark/README.md` --- - [ ] Yes (explain below) - [x] No No configured image defaults, chart values, resource names, ports, Secrets, ConfigMaps, or installation behavior are changed. --- The Spark 4 images intentionally exclude S3A, ABFS, GCS, and BigQuery connectors, so connector support requires a later immutable image revision. JFrog publication uses immutable tags and remains a manual checkpoint; attach the resulting repository digests, pull-by-digest output, image inspection output, version output, JAR inventories, and source commit to ML-13080. - Confirm both immutable JFrog images were published successfully and attach the required evidence to ML-13080. - Installation on Docker Desktop and a real Kubernetes cluster still requires human confirmation. --- docker/spark/Dockerfile.spark4 | 27 +++++ docker/spark/Dockerfile.spark4.cuda | 80 ++++++++++++++ docker/spark/Makefile | 31 ++++++ docker/spark/README.md | 82 ++++++++++++++ docker/spark/scripts/ce-customize-spark4.sh | 37 +++++++ docker/spark/scripts/jar-parity.sh | 46 ++++++++ docker/spark/scripts/validate-spark4.sh | 113 ++++++++++++++++++++ 7 files changed, 416 insertions(+) create mode 100644 docker/spark/Dockerfile.spark4 create mode 100644 docker/spark/Dockerfile.spark4.cuda create mode 100644 docker/spark/scripts/ce-customize-spark4.sh create mode 100644 docker/spark/scripts/jar-parity.sh create mode 100644 docker/spark/scripts/validate-spark4.sh diff --git a/docker/spark/Dockerfile.spark4 b/docker/spark/Dockerfile.spark4 new file mode 100644 index 00000000..7fb25ec6 --- /dev/null +++ b/docker/spark/Dockerfile.spark4 @@ -0,0 +1,27 @@ +# Copyright 2026 Iguazio +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# +# Spark 4 counterpart of ./Dockerfile. +# Spark 4.2.0, Scala 2.13, Temurin JDK 25, Hadoop 3.5.0 as bundled by the +# pinned upstream image. + +FROM spark@sha256:66e39dccde81909c23e5c56f4b465db6de2bdc38cf569aacc9e2380ee5005440 + +USER root + +# Shared with the Spark 4 CUDA image. +COPY scripts/ce-customize-spark4.sh /tmp/ce-customize-spark4.sh +RUN chmod +x /tmp/ce-customize-spark4.sh && /tmp/ce-customize-spark4.sh && rm -f /tmp/ce-customize-spark4.sh + +USER spark diff --git a/docker/spark/Dockerfile.spark4.cuda b/docker/spark/Dockerfile.spark4.cuda new file mode 100644 index 00000000..ff55941e --- /dev/null +++ b/docker/spark/Dockerfile.spark4.cuda @@ -0,0 +1,80 @@ +# Copyright 2026 Iguazio +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# +# CUDA counterpart of ./Dockerfile.spark4. The Spark distribution, the JDK, +# and the entrypoint are copied from the same digest-pinned base image the +# regular Spark 4 image is built from, so the two variants cannot drift and +# the build does not download Spark or the JDK a second time. + +ARG SPARK4_BASE_IMAGE=spark@sha256:66e39dccde81909c23e5c56f4b465db6de2bdc38cf569aacc9e2380ee5005440 +ARG SPARK4_CUDA_BASE_IMAGE=nvidia/cuda@sha256:61f6c08f2b59036cb935e56d1e31a6b64e3ae2c7ddb86d33fa0b044c7917b719 + +FROM ${SPARK4_BASE_IMAGE} AS spark4 + +FROM ${SPARK4_CUDA_BASE_IMAGE} + +# CUDA_VERSION and NV_CUDNN_VERSION are the CUDA base image's own environment. +# Do not declare an ARG of either name here: it would shadow the inherited +# value and let the labels describe a base this image was not built on. +LABEL com.iguazio.cuda-version="${CUDA_VERSION}" \ + com.iguazio.cudnn-version="${NV_CUDNN_VERSION}" + +ARG spark_uid=185 + +RUN groupadd --system --gid=${spark_uid} spark && \ + useradd --system --uid=${spark_uid} --gid=spark -d /nonexistent spark + +# Packages the upstream Spark image provides and the CUDA base does not. +# `locales` is one of them: the CUDA base generates only C and C.utf8, so +# en_US.UTF-8 has to be generated here for the locale ENV below to be valid. +RUN set -ex; \ + export DEBIAN_FRONTEND=noninteractive; \ + apt-get update; \ + apt-get install -y --no-install-recommends \ + gnupg2 bash tini libc6 libpam-modules \ + krb5-user libnss3 procps net-tools gosu libnss-wrapper libjemalloc2 \ + locales; \ + locale-gen en_US.UTF-8; \ + echo "auth required pam_wheel.so use_uid" >> /etc/pam.d/su; \ + rm -rf /var/lib/apt/lists/* + +# COPY carries no image environment, so everything the Spark base sets and +# this image needs is restated here. +ENV LANG=en_US.UTF-8 \ + LANGUAGE=en_US:en \ + LC_ALL=en_US.UTF-8 + +COPY --from=spark4 /opt/java/openjdk /opt/java/openjdk +COPY --from=spark4 /opt/spark /opt/spark +COPY --from=spark4 /opt/decom.sh /opt/decom.sh +COPY --from=spark4 /opt/entrypoint.sh /opt/entrypoint.sh + +# COPY creates the destination directory itself as root, while its contents +# keep the source image's ownership. Restore the Spark base's owner on the +# directory so the two variants match exactly. +RUN chown spark:spark /opt/spark + +ENV JAVA_HOME=/opt/java/openjdk \ + SPARK_HOME=/opt/spark +ENV PATH="${JAVA_HOME}/bin:${PATH}" + +WORKDIR /opt/spark/work-dir + +# Shared with the regular Spark 4 image. +COPY scripts/ce-customize-spark4.sh /tmp/ce-customize-spark4.sh +RUN chmod +x /tmp/ce-customize-spark4.sh && /tmp/ce-customize-spark4.sh && rm -f /tmp/ce-customize-spark4.sh + +USER spark + +ENTRYPOINT [ "/opt/entrypoint.sh" ] diff --git a/docker/spark/Makefile b/docker/spark/Makefile index 3c515742..ed1a54d5 100644 --- a/docker/spark/Makefile +++ b/docker/spark/Makefile @@ -27,3 +27,34 @@ validate-cuda: .PHONY: validate-all validate-all: validate validate-cuda + +# Spark 4 variables and targets. +REGISTRY ?= gcr.io/iguazio +MLRUN_CE_SPARK4_IMAGE_TAG ?= $(REGISTRY)/spark-app:4.2.0-scala2.13-java25-ubuntu-1 +MLRUN_CE_SPARK4_CUDA_IMAGE_TAG ?= $(REGISTRY)/spark-app-cuda:4.2.0-scala2.13-java25-ubuntu-1 + +.PHONY: build-spark4 +build-spark4: + docker build --platform="$(MLRUN_CE_IMAGE_PLATFORM)" -t "$(MLRUN_CE_SPARK4_IMAGE_TAG)" -f Dockerfile.spark4 . + +.PHONY: build-spark4-cuda +build-spark4-cuda: + docker build --platform="$(MLRUN_CE_IMAGE_PLATFORM)" -t "$(MLRUN_CE_SPARK4_CUDA_IMAGE_TAG)" -f Dockerfile.spark4.cuda . + +.PHONY: build-spark4-all +build-spark4-all: build-spark4 build-spark4-cuda + +.PHONY: validate-spark4 +validate-spark4: + bash scripts/validate-spark4.sh "$(MLRUN_CE_SPARK4_IMAGE_TAG)" + +.PHONY: validate-spark4-cuda +validate-spark4-cuda: + bash scripts/validate-spark4.sh "$(MLRUN_CE_SPARK4_CUDA_IMAGE_TAG)" --cuda + +.PHONY: validate-spark4-all +validate-spark4-all: validate-spark4 validate-spark4-cuda + +.PHONY: jar-parity-spark4 +jar-parity-spark4: + bash scripts/jar-parity.sh "$(MLRUN_CE_SPARK4_IMAGE_TAG)" "$(MLRUN_CE_SPARK4_CUDA_IMAGE_TAG)" diff --git a/docker/spark/README.md b/docker/spark/README.md index 3bc3d8be..24e47a69 100644 --- a/docker/spark/README.md +++ b/docker/spark/README.md @@ -49,3 +49,85 @@ docker inspect --format '{{index .RepoDigests 0}}' \ Do not republish the existing regular image. Record the CUDA image digest and the CE source commit. + +## Spark 4 + +The Spark 4 images use Spark 4.2.0, Scala 2.13, Hadoop 3.5.0, Temurin +25.0.4+7, and Python 3.11: + +- `spark-app:4.2.0-scala2.13-java25-ubuntu-1` uses + `spark@sha256:66e39dccde81909c23e5c56f4b465db6de2bdc38cf569aacc9e2380ee5005440`. +- `spark-app-cuda:4.2.0-scala2.13-java25-ubuntu-1` copies Spark and Java from + that image into + `nvidia/cuda@sha256:61f6c08f2b59036cb935e56d1e31a6b64e3ae2c7ddb86d33fa0b044c7917b719` + (CUDA 12.8.1, cuDNN 9.8.0.87-1). + +These images do not include S3A, ABFS, GCS, or BigQuery connectors. They +therefore cannot access SeaweedFS through S3A. Connector support requires a +new immutable image revision, such as +`4.2.0-scala2.13-java25-ubuntu-2`. + +MLRun selects the repository and tag through `MLRUN_SPARK_APP_IMAGE` and +`MLRUN_SPARK_APP_IMAGE_TAG`. mlefi resolves the `spark-app` / +`spark-app-cuda` pair using `^(.+?)-scala.*$`. These recipes do not change +the configured defaults. + +### Build + +```bash +make build-spark4 +make build-spark4-cuda +make build-spark4-all +``` + +`REGISTRY` defaults to `gcr.io/iguazio`. Override it or either image-tag +variable as needed. + +### Validate + +```bash +make validate-spark4-all +make jar-parity-spark4 +``` + +The validators check Spark, Scala, Java, Hadoop, Python, image metadata, and +CUDA metadata. The parity check compares every `$SPARK_HOME/jars` entry by +SHA-256. + +### Publish (manual, JFrog) + +Publication is manual. First check that neither +`spark-app:4.2.0-scala2.13-java25-ubuntu-1` nor +`spark-app-cuda:4.2.0-scala2.13-java25-ubuntu-1` already exists in +`mckinsey-ig4-next-gen-docker-local.jfrog.io`. Never overwrite an existing +immutable tag. + +```bash +make REGISTRY=mckinsey-ig4-next-gen-docker-local.jfrog.io build-spark4-all +make REGISTRY=mckinsey-ig4-next-gen-docker-local.jfrog.io validate-spark4-all +make REGISTRY=mckinsey-ig4-next-gen-docker-local.jfrog.io jar-parity-spark4 + +docker login mckinsey-ig4-next-gen-docker-local.jfrog.io +docker push mckinsey-ig4-next-gen-docker-local.jfrog.io/spark-app:4.2.0-scala2.13-java25-ubuntu-1 +docker push mckinsey-ig4-next-gen-docker-local.jfrog.io/spark-app-cuda:4.2.0-scala2.13-java25-ubuntu-1 + +docker inspect --format '{{index .RepoDigests 0}}' \ + mckinsey-ig4-next-gen-docker-local.jfrog.io/spark-app:4.2.0-scala2.13-java25-ubuntu-1 +docker inspect --format '{{index .RepoDigests 0}}' \ + mckinsey-ig4-next-gen-docker-local.jfrog.io/spark-app-cuda:4.2.0-scala2.13-java25-ubuntu-1 +``` + +Record both repository digests, then pull each image back by digest and +rerun validation against the digest references. A local image ID is not a +published digest. + +### Jira evidence (ML-13080) + +Attach to ML-13080: + +- immutable tags and repository digests; +- pull-by-digest and image-inspection output; +- `spark-submit --version` and `java -version` output; +- CPU and CUDA JAR SHA-256 inventories and the parity result; +- the source commit; +- the connector limitations documented above. diff --git a/docker/spark/scripts/ce-customize-spark4.sh b/docker/spark/scripts/ce-customize-spark4.sh new file mode 100644 index 00000000..cc276cfa --- /dev/null +++ b/docker/spark/scripts/ce-customize-spark4.sh @@ -0,0 +1,37 @@ +#!/bin/bash +# Copyright 2026 Iguazio +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# +# Spark 4 counterpart of ce-customize.sh. Installs Python 3.11 and gives the +# spark user a writable home directory. Installs no cloud connector JARs. +set -ex +export DEBIAN_FRONTEND=noninteractive + +apt-get update +apt-get install -y --no-install-recommends software-properties-common curl ca-certificates gnupg +add-apt-repository -y ppa:deadsnakes/ppa +apt-get update +apt-get install -y python3.11 python3.11-distutils git +rm -rf /var/lib/apt/lists/* + +curl https://bootstrap.pypa.io/get-pip.py -o /tmp/get-pip.py +python3.11 /tmp/get-pip.py +rm -f /tmp/get-pip.py + +update-alternatives --install /usr/bin/python3 python3 /usr/bin/python3.11 1 +ln -sf /usr/bin/python3 /usr/bin/python + +usermod -d /home/spark spark +mkdir -p /home/spark +chown spark:spark /home/spark diff --git a/docker/spark/scripts/jar-parity.sh b/docker/spark/scripts/jar-parity.sh new file mode 100644 index 00000000..de6a8b45 --- /dev/null +++ b/docker/spark/scripts/jar-parity.sh @@ -0,0 +1,46 @@ +#!/bin/bash + +# Copyright 2026 Iguazio +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# +# Asserts that two images carry the same $SPARK_HOME/jars by content, not just +# by filename. + +set -euo pipefail + +IMAGE_A="${1:?usage: jar-parity.sh }" +IMAGE_B="${2:?usage: jar-parity.sh }" + +checksums() { + # LC_ALL=C: the two images may differ in locale, and collation order would + # otherwise diff even when the contents match. + docker run --rm --platform "${MLRUN_CE_IMAGE_PLATFORM:-linux/amd64}" \ + --entrypoint bash "$1" -c \ + 'cd "$SPARK_HOME/jars" && LC_ALL=C sha256sum *.jar | LC_ALL=C sort' +} + +echo "==> comparing \$SPARK_HOME/jars: $IMAGE_A vs $IMAGE_B" +a="$(checksums "$IMAGE_A")" +b="$(checksums "$IMAGE_B")" + +count="$(wc -l <<<"$a" | tr -d '[:space:]')" +[[ "$count" -gt 0 ]] || { echo "FAIL: no JARs found in $IMAGE_A"; exit 1; } + +if [[ "$a" != "$b" ]]; then + echo "FAIL: JAR contents differ:" + diff <(echo "$a") <(echo "$b") || true + exit 1 +fi + +echo "==> $count JARs identical by SHA-256" diff --git a/docker/spark/scripts/validate-spark4.sh b/docker/spark/scripts/validate-spark4.sh new file mode 100644 index 00000000..40885b50 --- /dev/null +++ b/docker/spark/scripts/validate-spark4.sh @@ -0,0 +1,113 @@ +#!/bin/bash + +# Copyright 2026 Iguazio +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# +# Spark 4 counterpart of validate.sh. Local image checks that do not require +# a GPU. + +set -euo pipefail + +IMAGE="${1:?usage: validate-spark4.sh [--cuda]}" +CUDA_MODE="${2:-}" + +PINNED_CUDA_VERSION="12.8.1" +PINNED_CUDNN_VERSION="9.8.0.87-1" + +EXPECTED_JARS=( + hadoop-client-api-3.5.0.jar + hadoop-client-runtime-3.5.0.jar +) + +run() { + docker run --rm --platform "${MLRUN_CE_IMAGE_PLATFORM:-linux/amd64}" --entrypoint bash "$IMAGE" -c "$1" +} + +echo "==> [$IMAGE] linux/amd64 architecture" +arch="$(docker inspect "$IMAGE" --format '{{.Os}}/{{.Architecture}}')" +[[ "$arch" == "linux/amd64" ]] || { echo "FAIL: unexpected architecture: '$arch'"; exit 1; } +uname_m="$(run 'uname -m')" +[[ "$uname_m" == "x86_64" ]] || { echo "FAIL: unexpected uname -m: '$uname_m'"; exit 1; } + +echo "==> [$IMAGE] preserves the Spark entrypoint" +entrypoint="$(docker inspect "$IMAGE" --format '{{json .Config.Entrypoint}}')" +[[ "$entrypoint" == '["/opt/entrypoint.sh"]' ]] || { echo "FAIL: unexpected entrypoint: '$entrypoint'"; exit 1; } + +echo "==> [$IMAGE] runs as the spark user" +whoami="$(run 'whoami')" +[[ "$whoami" == "spark" ]] || { echo "FAIL: expected spark user, got '$whoami'"; exit 1; } + +echo "==> [$IMAGE] working spark-submit / Spark 4.2.0 / Scala 2.13" +spark_version="$(run '$SPARK_HOME/bin/spark-submit --version 2>&1')" +grep -q 'version 4.2.0' <<<"$spark_version" || { echo "FAIL: Spark is not version 4.2.0:"; echo "$spark_version"; exit 1; } +grep -q 'Scala version 2.13' <<<"$spark_version" || { echo "FAIL: Scala is not version 2.13:"; echo "$spark_version"; exit 1; } + +echo "==> [$IMAGE] Temurin 25.0.4+7" +java_version="$(run 'java -version 2>&1')" +grep -Fq 'Temurin-25.0.4+7' <<<"$java_version" || { echo "FAIL: Java is not Temurin 25.0.4+7:"; echo "$java_version"; exit 1; } + +echo "==> [$IMAGE] Hadoop 3.5.0" +for jar in "${EXPECTED_JARS[@]}"; do + run "test -f \$SPARK_HOME/jars/$jar" || { echo "FAIL: missing jar $jar"; exit 1; } +done + +echo "==> [$IMAGE] Python 3.11" +python_version="$(run 'python3 --version 2>&1')" +grep -q 'Python 3.11' <<<"$python_version" || { echo "FAIL: Python is not version 3.11: $python_version"; exit 1; } + +echo "==> [$IMAGE] spark home directory is writable" +passwd_entry="$(run 'getent passwd spark')" +grep -q ':/home/spark:' <<<"$passwd_entry" || { echo "FAIL: spark home is not /home/spark: '$passwd_entry'"; exit 1; } +run 'test -w /home/spark' || { echo "FAIL: /home/spark is not writable by spark"; exit 1; } + +echo "==> [$IMAGE] \$SPARK_HOME ownership" +spark_home_owner="$(run 'stat -c %U:%G $SPARK_HOME')" +[[ "$spark_home_owner" == "spark:spark" ]] || { echo "FAIL: unexpected \$SPARK_HOME ownership: '$spark_home_owner'"; exit 1; } + +echo "==> [$IMAGE] UTF-8 locale" +charmap="$(run 'locale charmap')" +[[ "$charmap" == "UTF-8" ]] || { echo "FAIL: expected UTF-8, got '$charmap'"; exit 1; } + +if [[ "$CUDA_MODE" == "--cuda" ]]; then + echo "==> [$IMAGE] CUDA_VERSION=$PINNED_CUDA_VERSION" + cuda_env="$(docker inspect "$IMAGE" --format '{{range .Config.Env}}{{println .}}{{end}}' | grep '^CUDA_VERSION=' || true)" + [[ "$cuda_env" == "CUDA_VERSION=$PINNED_CUDA_VERSION" ]] || { echo "FAIL: unexpected CUDA_VERSION: '$cuda_env'"; exit 1; } + + echo "==> [$IMAGE] NV_CUDNN_VERSION=$PINNED_CUDNN_VERSION" + cudnn_env="$(docker inspect "$IMAGE" --format '{{range .Config.Env}}{{println .}}{{end}}' | grep '^NV_CUDNN_VERSION=' || true)" + [[ "$cudnn_env" == "NV_CUDNN_VERSION=$PINNED_CUDNN_VERSION" ]] || { echo "FAIL: unexpected NV_CUDNN_VERSION: '$cudnn_env'"; exit 1; } + + echo "==> [$IMAGE] com.iguazio.cuda-version / com.iguazio.cudnn-version labels" + cuda_label="$(docker inspect "$IMAGE" --format '{{index .Config.Labels "com.iguazio.cuda-version"}}')" + [[ "$cuda_label" == "$PINNED_CUDA_VERSION" ]] || { echo "FAIL: unexpected com.iguazio.cuda-version label: '$cuda_label'"; exit 1; } + cudnn_label="$(docker inspect "$IMAGE" --format '{{index .Config.Labels "com.iguazio.cudnn-version"}}')" + [[ "$cudnn_label" == "$PINNED_CUDNN_VERSION" ]] || { echo "FAIL: unexpected com.iguazio.cudnn-version label: '$cudnn_label'"; exit 1; } + + echo "==> [$IMAGE] nvcc release 12.8" + nvcc_version="$(run 'nvcc --version 2>&1')" + grep -Fq 'release 12.8' <<<"$nvcc_version" || { echo "FAIL: CUDA toolkit is not 12.8: $nvcc_version"; exit 1; } + + echo "==> [$IMAGE] cuDNN 9 libraries present" + run 'ldconfig -p | grep -q libcudnn.so.9' || { echo "FAIL: libcudnn.so.9 not found"; exit 1; } + + echo "==> [$IMAGE] NVIDIA_VISIBLE_DEVICES=all" + visible_devices="$(docker inspect "$IMAGE" --format '{{range .Config.Env}}{{println .}}{{end}}' | grep '^NVIDIA_VISIBLE_DEVICES=' || true)" + [[ "$visible_devices" == "NVIDIA_VISIBLE_DEVICES=all" ]] || { echo "FAIL: unexpected NVIDIA_VISIBLE_DEVICES: '$visible_devices'"; exit 1; } + + echo "==> [$IMAGE] NVIDIA_DRIVER_CAPABILITIES=compute,utility" + caps="$(docker inspect "$IMAGE" --format '{{range .Config.Env}}{{println .}}{{end}}' | grep '^NVIDIA_DRIVER_CAPABILITIES=' || true)" + [[ "$caps" == "NVIDIA_DRIVER_CAPABILITIES=compute,utility" ]] || { echo "FAIL: unexpected NVIDIA_DRIVER_CAPABILITIES: '$caps'"; exit 1; } +fi + +echo "==> [$IMAGE] all checks passed" From 639138bb266dbfb86168489270cbdd275d8b9134 Mon Sep 17 00:00:00 2001 From: Gal Topper Date: Tue, 22 Sep 2026 18:09:59 +0700 Subject: [PATCH 2/5] Enforce Spark 4 image configuration parity --- docker/spark/Dockerfile.spark4 | 4 +- docker/spark/Dockerfile.spark4.cuda | 11 ++- docker/spark/Makefile | 17 +++- docker/spark/README.md | 31 ++++--- docker/spark/scripts/env-parity.sh | 106 ++++++++++++++++++++++++ docker/spark/scripts/jar-parity.sh | 2 +- docker/spark/scripts/validate-spark4.sh | 2 +- 7 files changed, 146 insertions(+), 27 deletions(-) create mode 100644 docker/spark/scripts/env-parity.sh diff --git a/docker/spark/Dockerfile.spark4 b/docker/spark/Dockerfile.spark4 index 7fb25ec6..4aeae348 100644 --- a/docker/spark/Dockerfile.spark4 +++ b/docker/spark/Dockerfile.spark4 @@ -1,3 +1,4 @@ +# check=skip=InvalidDefaultArgInFrom # Copyright 2026 Iguazio # # Licensed under the Apache License, Version 2.0 (the "License"); @@ -16,7 +17,8 @@ # Spark 4.2.0, Scala 2.13, Temurin JDK 25, Hadoop 3.5.0 as bundled by the # pinned upstream image. -FROM spark@sha256:66e39dccde81909c23e5c56f4b465db6de2bdc38cf569aacc9e2380ee5005440 +ARG SPARK4_BASE_IMAGE +FROM ${SPARK4_BASE_IMAGE} USER root diff --git a/docker/spark/Dockerfile.spark4.cuda b/docker/spark/Dockerfile.spark4.cuda index ff55941e..ac2bf3f1 100644 --- a/docker/spark/Dockerfile.spark4.cuda +++ b/docker/spark/Dockerfile.spark4.cuda @@ -1,3 +1,4 @@ +# check=skip=InvalidDefaultArgInFrom # Copyright 2026 Iguazio # # Licensed under the Apache License, Version 2.0 (the "License"); @@ -17,8 +18,8 @@ # regular Spark 4 image is built from, so the two variants cannot drift and # the build does not download Spark or the JDK a second time. -ARG SPARK4_BASE_IMAGE=spark@sha256:66e39dccde81909c23e5c56f4b465db6de2bdc38cf569aacc9e2380ee5005440 -ARG SPARK4_CUDA_BASE_IMAGE=nvidia/cuda@sha256:61f6c08f2b59036cb935e56d1e31a6b64e3ae2c7ddb86d33fa0b044c7917b719 +ARG SPARK4_BASE_IMAGE +ARG SPARK4_CUDA_BASE_IMAGE FROM ${SPARK4_BASE_IMAGE} AS spark4 @@ -49,8 +50,9 @@ RUN set -ex; \ echo "auth required pam_wheel.so use_uid" >> /etc/pam.d/su; \ rm -rf /var/lib/apt/lists/* -# COPY carries no image environment, so everything the Spark base sets and -# this image needs is restated here. +# COPY carries no image environment, so the runtime settings from the Spark +# base are restated here. SPARK_TGZ_URL, SPARK_TGZ_ASC_URL, and GPG_KEY are +# build-provenance metadata and are intentionally omitted. ENV LANG=en_US.UTF-8 \ LANGUAGE=en_US:en \ LC_ALL=en_US.UTF-8 @@ -66,6 +68,7 @@ COPY --from=spark4 /opt/entrypoint.sh /opt/entrypoint.sh RUN chown spark:spark /opt/spark ENV JAVA_HOME=/opt/java/openjdk \ + JAVA_VERSION=jdk-25.0.4+7 \ SPARK_HOME=/opt/spark ENV PATH="${JAVA_HOME}/bin:${PATH}" diff --git a/docker/spark/Makefile b/docker/spark/Makefile index ed1a54d5..bde6944d 100644 --- a/docker/spark/Makefile +++ b/docker/spark/Makefile @@ -30,16 +30,23 @@ validate-all: validate validate-cuda # Spark 4 variables and targets. REGISTRY ?= gcr.io/iguazio +SPARK4_BASE_IMAGE ?= spark@sha256:66e39dccde81909c23e5c56f4b465db6de2bdc38cf569aacc9e2380ee5005440 +SPARK4_CUDA_BASE_IMAGE ?= nvidia/cuda@sha256:61f6c08f2b59036cb935e56d1e31a6b64e3ae2c7ddb86d33fa0b044c7917b719 MLRUN_CE_SPARK4_IMAGE_TAG ?= $(REGISTRY)/spark-app:4.2.0-scala2.13-java25-ubuntu-1 MLRUN_CE_SPARK4_CUDA_IMAGE_TAG ?= $(REGISTRY)/spark-app-cuda:4.2.0-scala2.13-java25-ubuntu-1 .PHONY: build-spark4 build-spark4: - docker build --platform="$(MLRUN_CE_IMAGE_PLATFORM)" -t "$(MLRUN_CE_SPARK4_IMAGE_TAG)" -f Dockerfile.spark4 . + docker build --platform=linux/amd64 \ + --build-arg SPARK4_BASE_IMAGE="$(SPARK4_BASE_IMAGE)" \ + -t "$(MLRUN_CE_SPARK4_IMAGE_TAG)" -f Dockerfile.spark4 . .PHONY: build-spark4-cuda build-spark4-cuda: - docker build --platform="$(MLRUN_CE_IMAGE_PLATFORM)" -t "$(MLRUN_CE_SPARK4_CUDA_IMAGE_TAG)" -f Dockerfile.spark4.cuda . + docker build --platform=linux/amd64 \ + --build-arg SPARK4_BASE_IMAGE="$(SPARK4_BASE_IMAGE)" \ + --build-arg SPARK4_CUDA_BASE_IMAGE="$(SPARK4_CUDA_BASE_IMAGE)" \ + -t "$(MLRUN_CE_SPARK4_CUDA_IMAGE_TAG)" -f Dockerfile.spark4.cuda . .PHONY: build-spark4-all build-spark4-all: build-spark4 build-spark4-cuda @@ -53,8 +60,12 @@ validate-spark4-cuda: bash scripts/validate-spark4.sh "$(MLRUN_CE_SPARK4_CUDA_IMAGE_TAG)" --cuda .PHONY: validate-spark4-all -validate-spark4-all: validate-spark4 validate-spark4-cuda +validate-spark4-all: validate-spark4 validate-spark4-cuda jar-parity-spark4 env-parity-spark4 .PHONY: jar-parity-spark4 jar-parity-spark4: bash scripts/jar-parity.sh "$(MLRUN_CE_SPARK4_IMAGE_TAG)" "$(MLRUN_CE_SPARK4_CUDA_IMAGE_TAG)" + +.PHONY: env-parity-spark4 +env-parity-spark4: + bash scripts/env-parity.sh "$(MLRUN_CE_SPARK4_IMAGE_TAG)" "$(MLRUN_CE_SPARK4_CUDA_IMAGE_TAG)" diff --git a/docker/spark/README.md b/docker/spark/README.md index 24e47a69..013f8e0f 100644 --- a/docker/spark/README.md +++ b/docker/spark/README.md @@ -52,25 +52,24 @@ the CE source commit. ## Spark 4 -The Spark 4 images use Spark 4.2.0, Scala 2.13, Hadoop 3.5.0, Temurin -25.0.4+7, and Python 3.11: +The Spark 4 images target `linux/amd64` and use Spark 4.2.0, Scala 2.13, +Hadoop 3.5.0, Temurin 25.0.4+7, and Python 3.11: -- `spark-app:4.2.0-scala2.13-java25-ubuntu-1` uses - `spark@sha256:66e39dccde81909c23e5c56f4b465db6de2bdc38cf569aacc9e2380ee5005440`. +- `spark-app:4.2.0-scala2.13-java25-ubuntu-1` uses the digest-pinned Spark + base configured by `SPARK4_BASE_IMAGE` in the Makefile. - `spark-app-cuda:4.2.0-scala2.13-java25-ubuntu-1` copies Spark and Java from - that image into - `nvidia/cuda@sha256:61f6c08f2b59036cb935e56d1e31a6b64e3ae2c7ddb86d33fa0b044c7917b719` - (CUDA 12.8.1, cuDNN 9.8.0.87-1). + that base into the digest-pinned `SPARK4_CUDA_BASE_IMAGE` (CUDA 12.8.1, + cuDNN 9.8.0.87-1). These images do not include S3A, ABFS, GCS, or BigQuery connectors. They therefore cannot access SeaweedFS through S3A. Connector support requires a new immutable image revision, such as `4.2.0-scala2.13-java25-ubuntu-2`. -MLRun selects the repository and tag through `MLRUN_SPARK_APP_IMAGE` and -`MLRUN_SPARK_APP_IMAGE_TAG`. mlefi resolves the `spark-app` / -`spark-app-cuda` pair using `^(.+?)-scala.*$`. These recipes do not change -the configured defaults. +MLRun selects the CPU repository and tag through `MLRUN_SPARK_APP_IMAGE` and +`MLRUN_SPARK_APP_IMAGE_TAG`, and derives the CUDA repository by appending +`-cuda`. mlefi recognizes that resulting `spark-app` / `spark-app-cuda` pair +using `^(.+?)-scala.*$`. These recipes do not change the configured defaults. ### Build @@ -80,19 +79,18 @@ make build-spark4-cuda make build-spark4-all ``` -`REGISTRY` defaults to `gcr.io/iguazio`. Override it or either image-tag -variable as needed. +`REGISTRY` defaults to `gcr.io/iguazio`. The Makefile is the source of truth +for both pinned base-image digests. ### Validate ```bash make validate-spark4-all -make jar-parity-spark4 ``` The validators check Spark, Scala, Java, Hadoop, Python, image metadata, and -CUDA metadata. The parity check compares every `$SPARK_HOME/jars` entry by -SHA-256. +CUDA metadata. The aggregate target also checks CPU/CUDA environment parity +and compares every `$SPARK_HOME/jars` entry by SHA-256. ### Publish (manual, JFrog) @@ -105,7 +103,6 @@ immutable tag. ```bash make REGISTRY=mckinsey-ig4-next-gen-docker-local.jfrog.io build-spark4-all make REGISTRY=mckinsey-ig4-next-gen-docker-local.jfrog.io validate-spark4-all -make REGISTRY=mckinsey-ig4-next-gen-docker-local.jfrog.io jar-parity-spark4 docker login mckinsey-ig4-next-gen-docker-local.jfrog.io docker push mckinsey-ig4-next-gen-docker-local.jfrog.io/spark-app:4.2.0-scala2.13-java25-ubuntu-1 diff --git a/docker/spark/scripts/env-parity.sh b/docker/spark/scripts/env-parity.sh new file mode 100644 index 00000000..3ba33298 --- /dev/null +++ b/docker/spark/scripts/env-parity.sh @@ -0,0 +1,106 @@ +#!/bin/bash + +# Copyright 2026 Iguazio +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# +# Compares CPU and CUDA image environments. CUDA runtime variables and the +# intentional omission of Spark build-provenance variables are allowed. + +set -euo pipefail + +CPU_IMAGE="${1:?usage: env-parity.sh }" +CUDA_IMAGE="${2:?usage: env-parity.sh }" + +image_env() { + docker inspect "$1" --format '{{range .Config.Env}}{{println .}}{{end}}' +} + +lookup() { + local env_text="$1" + local key="$2" + local line + + while IFS= read -r line; do + if [[ "${line%%=*}" == "$key" ]]; then + printf '%s\n' "${line#*=}" + return 0 + fi + done <<<"$env_text" + return 1 +} + +cpu_env="$(image_env "$CPU_IMAGE")" +cuda_env="$(image_env "$CUDA_IMAGE")" + +echo "==> comparing shared image environment" +while IFS= read -r line; do + [[ -n "$line" ]] || continue + key="${line%%=*}" + cpu_value="${line#*=}" + + case "$key" in + SPARK_TGZ_URL|SPARK_TGZ_ASC_URL|GPG_KEY) + if lookup "$cuda_env" "$key" >/dev/null; then + echo "FAIL: CUDA image unexpectedly contains $key" + exit 1 + fi + ;; + PATH) + cuda_value="$(lookup "$cuda_env" "$key")" || { + echo "FAIL: CUDA image is missing $key" + exit 1 + } + normalized_cuda_path="${cuda_value/:\/usr\/local\/cuda\/bin/}" + [[ "$normalized_cuda_path" == "$cpu_value" ]] || { + echo "FAIL: PATH differs beyond the expected CUDA addition" + echo "CPU: $cpu_value" + echo "CUDA: $cuda_value" + exit 1 + } + ;; + *) + cuda_value="$(lookup "$cuda_env" "$key")" || { + echo "FAIL: CUDA image is missing $key" + exit 1 + } + [[ "$cuda_value" == "$cpu_value" ]] || { + echo "FAIL: $key differs" + echo "CPU: $cpu_value" + echo "CUDA: $cuda_value" + exit 1 + } + ;; + esac +done <<<"$cpu_env" + +echo "==> checking CUDA-only environment allowlist" +while IFS= read -r line; do + [[ -n "$line" ]] || continue + key="${line%%=*}" + + if lookup "$cpu_env" "$key" >/dev/null; then + continue + fi + + case "$key" in + NV_*|CUDA_*|NVIDIA_*|NVARCH|NCCL_VERSION|LD_LIBRARY_PATH|LIBRARY_PATH) + ;; + *) + echo "FAIL: unexpected CUDA-only environment variable: $key" + exit 1 + ;; + esac +done <<<"$cuda_env" + +echo "==> CPU/CUDA environment parity passed" diff --git a/docker/spark/scripts/jar-parity.sh b/docker/spark/scripts/jar-parity.sh index de6a8b45..62d6d128 100644 --- a/docker/spark/scripts/jar-parity.sh +++ b/docker/spark/scripts/jar-parity.sh @@ -25,7 +25,7 @@ IMAGE_B="${2:?usage: jar-parity.sh }" checksums() { # LC_ALL=C: the two images may differ in locale, and collation order would # otherwise diff even when the contents match. - docker run --rm --platform "${MLRUN_CE_IMAGE_PLATFORM:-linux/amd64}" \ + docker run --rm --platform linux/amd64 \ --entrypoint bash "$1" -c \ 'cd "$SPARK_HOME/jars" && LC_ALL=C sha256sum *.jar | LC_ALL=C sort' } diff --git a/docker/spark/scripts/validate-spark4.sh b/docker/spark/scripts/validate-spark4.sh index 40885b50..f13c689e 100644 --- a/docker/spark/scripts/validate-spark4.sh +++ b/docker/spark/scripts/validate-spark4.sh @@ -31,7 +31,7 @@ EXPECTED_JARS=( ) run() { - docker run --rm --platform "${MLRUN_CE_IMAGE_PLATFORM:-linux/amd64}" --entrypoint bash "$IMAGE" -c "$1" + docker run --rm --platform linux/amd64 --entrypoint bash "$IMAGE" -c "$1" } echo "==> [$IMAGE] linux/amd64 architecture" From 9f537b33915937aaecfdbc20a3b7d5eb29b0f4e6 Mon Sep 17 00:00:00 2001 From: Gal Topper Date: Tue, 22 Sep 2026 18:15:32 +0700 Subject: [PATCH 3/5] Clarify Spark 4 image parity and selection --- docker/spark/Dockerfile.spark4 | 1 + docker/spark/Dockerfile.spark4.cuda | 1 + docker/spark/Makefile | 2 ++ docker/spark/README.md | 7 +++++-- docker/spark/scripts/env-parity.sh | 4 ++++ 5 files changed, 13 insertions(+), 2 deletions(-) diff --git a/docker/spark/Dockerfile.spark4 b/docker/spark/Dockerfile.spark4 index 4aeae348..6e6208cd 100644 --- a/docker/spark/Dockerfile.spark4 +++ b/docker/spark/Dockerfile.spark4 @@ -17,6 +17,7 @@ # Spark 4.2.0, Scala 2.13, Temurin JDK 25, Hadoop 3.5.0 as bundled by the # pinned upstream image. +# Required build argument; the Makefile supplies the pinned base reference. ARG SPARK4_BASE_IMAGE FROM ${SPARK4_BASE_IMAGE} diff --git a/docker/spark/Dockerfile.spark4.cuda b/docker/spark/Dockerfile.spark4.cuda index ac2bf3f1..945b641a 100644 --- a/docker/spark/Dockerfile.spark4.cuda +++ b/docker/spark/Dockerfile.spark4.cuda @@ -18,6 +18,7 @@ # regular Spark 4 image is built from, so the two variants cannot drift and # the build does not download Spark or the JDK a second time. +# Required build arguments; the Makefile supplies the pinned base references. ARG SPARK4_BASE_IMAGE ARG SPARK4_CUDA_BASE_IMAGE diff --git a/docker/spark/Makefile b/docker/spark/Makefile index bde6944d..994616af 100644 --- a/docker/spark/Makefile +++ b/docker/spark/Makefile @@ -30,6 +30,8 @@ validate-all: validate validate-cuda # Spark 4 variables and targets. REGISTRY ?= gcr.io/iguazio +# Spark uses a multi-platform index; CUDA uses its linux/amd64 child manifest +# because the Spark 4 image pair is intentionally amd64-only. SPARK4_BASE_IMAGE ?= spark@sha256:66e39dccde81909c23e5c56f4b465db6de2bdc38cf569aacc9e2380ee5005440 SPARK4_CUDA_BASE_IMAGE ?= nvidia/cuda@sha256:61f6c08f2b59036cb935e56d1e31a6b64e3ae2c7ddb86d33fa0b044c7917b719 MLRUN_CE_SPARK4_IMAGE_TAG ?= $(REGISTRY)/spark-app:4.2.0-scala2.13-java25-ubuntu-1 diff --git a/docker/spark/README.md b/docker/spark/README.md index 013f8e0f..0cc43b79 100644 --- a/docker/spark/README.md +++ b/docker/spark/README.md @@ -68,8 +68,10 @@ new immutable image revision, such as MLRun selects the CPU repository and tag through `MLRUN_SPARK_APP_IMAGE` and `MLRUN_SPARK_APP_IMAGE_TAG`, and derives the CUDA repository by appending -`-cuda`. mlefi recognizes that resulting `spark-app` / `spark-app-cuda` pair -using `^(.+?)-scala.*$`. These recipes do not change the configured defaults. +`-cuda`. mlefi resolves both repositories and extracts the Spark version from +the tag with `^(.+?)-scala.*$`; for example, +`4.2.0-scala2.13-java25-ubuntu-1` yields `4.2.0`. These recipes do not change +the configured defaults. ### Build @@ -126,5 +128,6 @@ Attach to ML-13080: - pull-by-digest and image-inspection output; - `spark-submit --version` and `java -version` output; - CPU and CUDA JAR SHA-256 inventories and the parity result; +- the pinned base-image references from the Makefile; - the source commit; - the connector limitations documented above. diff --git a/docker/spark/scripts/env-parity.sh b/docker/spark/scripts/env-parity.sh index 3ba33298..3355fa63 100644 --- a/docker/spark/scripts/env-parity.sh +++ b/docker/spark/scripts/env-parity.sh @@ -61,6 +61,10 @@ while IFS= read -r line; do echo "FAIL: CUDA image is missing $key" exit 1 } + [[ ":$cuda_value:" == *":/usr/local/cuda/bin:"* ]] || { + echo "FAIL: CUDA PATH is missing /usr/local/cuda/bin" + exit 1 + } normalized_cuda_path="${cuda_value/:\/usr\/local\/cuda\/bin/}" [[ "$normalized_cuda_path" == "$cpu_value" ]] || { echo "FAIL: PATH differs beyond the expected CUDA addition" From df3768d6cee67959b06df2ef6226cac0acb528a0 Mon Sep 17 00:00:00 2001 From: Gal Topper Date: Tue, 22 Sep 2026 19:35:39 +0700 Subject: [PATCH 4/5] Add cloud connectors to Spark 4 images --- docker/spark/Dockerfile.spark4 | 5 ++- docker/spark/Dockerfile.spark4.cuda | 5 ++- docker/spark/README.md | 35 ++++++++++++++++----- docker/spark/scripts/ce-customize-spark4.sh | 17 ++++++++-- docker/spark/scripts/jars-4.2.0.txt | 8 +++++ docker/spark/scripts/validate-spark4.sh | 35 +++++++++++++++++++-- 6 files changed, 91 insertions(+), 14 deletions(-) create mode 100644 docker/spark/scripts/jars-4.2.0.txt diff --git a/docker/spark/Dockerfile.spark4 b/docker/spark/Dockerfile.spark4 index 6e6208cd..4c825ab7 100644 --- a/docker/spark/Dockerfile.spark4 +++ b/docker/spark/Dockerfile.spark4 @@ -25,6 +25,9 @@ USER root # Shared with the Spark 4 CUDA image. COPY scripts/ce-customize-spark4.sh /tmp/ce-customize-spark4.sh -RUN chmod +x /tmp/ce-customize-spark4.sh && /tmp/ce-customize-spark4.sh && rm -f /tmp/ce-customize-spark4.sh +COPY scripts/jars-4.2.0.txt /tmp/jars-4.2.0.txt +RUN chmod +x /tmp/ce-customize-spark4.sh && \ + /tmp/ce-customize-spark4.sh /tmp/jars-4.2.0.txt && \ + rm -f /tmp/ce-customize-spark4.sh /tmp/jars-4.2.0.txt USER spark diff --git a/docker/spark/Dockerfile.spark4.cuda b/docker/spark/Dockerfile.spark4.cuda index 945b641a..57767845 100644 --- a/docker/spark/Dockerfile.spark4.cuda +++ b/docker/spark/Dockerfile.spark4.cuda @@ -77,7 +77,10 @@ WORKDIR /opt/spark/work-dir # Shared with the regular Spark 4 image. COPY scripts/ce-customize-spark4.sh /tmp/ce-customize-spark4.sh -RUN chmod +x /tmp/ce-customize-spark4.sh && /tmp/ce-customize-spark4.sh && rm -f /tmp/ce-customize-spark4.sh +COPY scripts/jars-4.2.0.txt /tmp/jars-4.2.0.txt +RUN chmod +x /tmp/ce-customize-spark4.sh && \ + /tmp/ce-customize-spark4.sh /tmp/jars-4.2.0.txt && \ + rm -f /tmp/ce-customize-spark4.sh /tmp/jars-4.2.0.txt USER spark diff --git a/docker/spark/README.md b/docker/spark/README.md index 0cc43b79..434d11c8 100644 --- a/docker/spark/README.md +++ b/docker/spark/README.md @@ -61,10 +61,27 @@ Hadoop 3.5.0, Temurin 25.0.4+7, and Python 3.11: that base into the digest-pinned `SPARK4_CUDA_BASE_IMAGE` (CUDA 12.8.1, cuDNN 9.8.0.87-1). -These images do not include S3A, ABFS, GCS, or BigQuery connectors. They -therefore cannot access SeaweedFS through S3A. Connector support requires a -new immutable image revision, such as -`4.2.0-scala2.13-java25-ubuntu-2`. +Both images install the connector artifacts listed in +`scripts/jars-4.2.0.txt`: Hadoop 3.5.0 connectors for S3A, ABFS, and GCS, +their required AWS and Azure dependencies, and the Scala 2.13 BigQuery +connector. These restore the connector family shipped by the published Spark +3 image so Spark 4 can serve as a like-for-like platform image. Local +validation checks packaging, class resolution, duplicate versions, and +CPU/CUDA parity. Authenticated provider testing and BigQuery compatibility +with Spark 4.2 are deferred to the corresponding activation work. + +`hadoop-gcp-3.5.0` provides +`org.apache.hadoop.fs.gs.GoogleHadoopFileSystem`, replacing the former +`com.google.cloud.hadoop.fs.gcs.GoogleHadoopFileSystem` class. It does not +provide a replacement `AbstractFileSystem` implementation. Consumers using +the old `fs.gs.impl` or `fs.AbstractFileSystem.gs.impl` configuration must +update or remove it. + +`hadoop-aws-3.5.0` uses AWS SDK v2. Hadoop remaps several common SDK v1 +credential-provider names, but not +`com.amazonaws.auth.DefaultAWSCredentialsProviderChain`, arbitrary +`com.amazonaws.*` providers, or custom SDK v1 implementations. Consumers +using those providers must migrate their configuration. MLRun selects the CPU repository and tag through `MLRUN_SPARK_APP_IMAGE` and `MLRUN_SPARK_APP_IMAGE_TAG`, and derives the CUDA repository by appending @@ -90,9 +107,10 @@ for both pinned base-image digests. make validate-spark4-all ``` -The validators check Spark, Scala, Java, Hadoop, Python, image metadata, and -CUDA metadata. The aggregate target also checks CPU/CUDA environment parity -and compares every `$SPARK_HOME/jars` entry by SHA-256. +The validators check Spark, Scala, Java, Hadoop, Python, the connector +inventory, image metadata, and CUDA metadata. The aggregate target also checks +CPU/CUDA environment parity and compares every `$SPARK_HOME/jars` entry by +SHA-256. It does not authenticate to AWS, Azure, or GCP services. ### Publish (manual, JFrog) @@ -130,4 +148,5 @@ Attach to ML-13080: - CPU and CUDA JAR SHA-256 inventories and the parity result; - the pinned base-image references from the Makefile; - the source commit; -- the connector limitations documented above. +- the connector inventory and deferred provider-validation status documented + above. diff --git a/docker/spark/scripts/ce-customize-spark4.sh b/docker/spark/scripts/ce-customize-spark4.sh index cc276cfa..f3a2e27e 100644 --- a/docker/spark/scripts/ce-customize-spark4.sh +++ b/docker/spark/scripts/ce-customize-spark4.sh @@ -13,11 +13,13 @@ # See the License for the specific language governing permissions and # limitations under the License. # -# Spark 4 counterpart of ce-customize.sh. Installs Python 3.11 and gives the -# spark user a writable home directory. Installs no cloud connector JARs. +# Spark 4 counterpart of ce-customize.sh. Installs Python 3.11, connector JARs +# from the supplied manifest, and a writable home directory for the spark user. set -ex export DEBIAN_FRONTEND=noninteractive +JAR_MANIFEST="${1:?usage: ce-customize-spark4.sh }" + apt-get update apt-get install -y --no-install-recommends software-properties-common curl ca-certificates gnupg add-apt-repository -y ppa:deadsnakes/ppa @@ -29,6 +31,17 @@ curl https://bootstrap.pypa.io/get-pip.py -o /tmp/get-pip.py python3.11 /tmp/get-pip.py rm -f /tmp/get-pip.py +while read -r line || [ -n "$line" ]; do + url="${line%%#*}" + url="$(echo "$url" | tr -d '[:space:]')" + [ -z "$url" ] && continue + case "$url" in + https://*) ;; + *) echo "bad manifest line: $line" >&2; exit 1 ;; + esac + curl -fsSL -o "/opt/spark/jars/$(basename "$url")" "$url" +done < "$JAR_MANIFEST" + update-alternatives --install /usr/bin/python3 python3 /usr/bin/python3.11 1 ln -sf /usr/bin/python3 /usr/bin/python diff --git a/docker/spark/scripts/jars-4.2.0.txt b/docker/spark/scripts/jars-4.2.0.txt new file mode 100644 index 00000000..1ed9b496 --- /dev/null +++ b/docker/spark/scripts/jars-4.2.0.txt @@ -0,0 +1,8 @@ +https://repo1.maven.org/maven2/org/apache/hadoop/hadoop-aws/3.5.0/hadoop-aws-3.5.0.jar +https://repo1.maven.org/maven2/software/amazon/s3/analyticsaccelerator/analyticsaccelerator-s3/1.3.1/analyticsaccelerator-s3-1.3.1.jar +https://repo1.maven.org/maven2/org/apache/hadoop/hadoop-azure/3.5.0/hadoop-azure-3.5.0.jar +https://repo1.maven.org/maven2/software/amazon/awssdk/bundle/2.35.4/bundle-2.35.4.jar +https://repo1.maven.org/maven2/org/wildfly/openssl/wildfly-openssl/2.2.5.Final/wildfly-openssl-2.2.5.Final.jar +https://repo1.maven.org/maven2/com/microsoft/azure/azure-storage/7.0.1/azure-storage-7.0.1.jar +https://repo1.maven.org/maven2/org/apache/hadoop/hadoop-gcp/3.5.0/hadoop-gcp-3.5.0.jar +https://repo1.maven.org/maven2/com/google/cloud/spark/spark-bigquery-with-dependencies_2.13/0.45.0/spark-bigquery-with-dependencies_2.13-0.45.0.jar diff --git a/docker/spark/scripts/validate-spark4.sh b/docker/spark/scripts/validate-spark4.sh index f13c689e..cd58e3ee 100644 --- a/docker/spark/scripts/validate-spark4.sh +++ b/docker/spark/scripts/validate-spark4.sh @@ -24,12 +24,24 @@ CUDA_MODE="${2:-}" PINNED_CUDA_VERSION="12.8.1" PINNED_CUDNN_VERSION="9.8.0.87-1" +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +JAR_MANIFEST="$SCRIPT_DIR/jars-4.2.0.txt" -EXPECTED_JARS=( +EXPECTED_BASE_JARS=( hadoop-client-api-3.5.0.jar hadoop-client-runtime-3.5.0.jar ) +[[ -f "$JAR_MANIFEST" ]] || { echo "missing JAR manifest: $JAR_MANIFEST" >&2; exit 1; } +EXPECTED_CONNECTOR_JARS=() +while read -r line || [ -n "$line" ]; do + url="${line%%#*}" + url="$(echo "$url" | tr -d '[:space:]')" + [ -z "$url" ] && continue + EXPECTED_CONNECTOR_JARS+=("$(basename "$url")") +done < "$JAR_MANIFEST" +[[ ${#EXPECTED_CONNECTOR_JARS[@]} -gt 0 ]] || { echo "no connector JARs listed in $JAR_MANIFEST" >&2; exit 1; } + run() { docker run --rm --platform linux/amd64 --entrypoint bash "$IMAGE" -c "$1" } @@ -58,10 +70,29 @@ java_version="$(run 'java -version 2>&1')" grep -Fq 'Temurin-25.0.4+7' <<<"$java_version" || { echo "FAIL: Java is not Temurin 25.0.4+7:"; echo "$java_version"; exit 1; } echo "==> [$IMAGE] Hadoop 3.5.0" -for jar in "${EXPECTED_JARS[@]}"; do +for jar in "${EXPECTED_BASE_JARS[@]}"; do run "test -f \$SPARK_HOME/jars/$jar" || { echo "FAIL: missing jar $jar"; exit 1; } done +echo "==> [$IMAGE] ${#EXPECTED_CONNECTOR_JARS[@]} connector JARs" +for jar in "${EXPECTED_CONNECTOR_JARS[@]}"; do + run "test -f \$SPARK_HOME/jars/$jar" || { echo "FAIL: missing connector jar $jar"; exit 1; } +done + +echo "==> [$IMAGE] no duplicate connector versions" +CONNECTOR_JAR_PREFIXES='^(hadoop-aws|hadoop-azure|hadoop-common|hadoop-gcp|aws-java-sdk-bundle|bundle|analyticsaccelerator-s3|wildfly-openssl|azure-storage|gcs-connector|jetty-util|jetty-util-ajax|spark-bigquery-with-dependencies)-' +duplicates="$(run 'ls "$SPARK_HOME/jars"' | grep -E "$CONNECTOR_JAR_PREFIXES" | sed -E 's/-[0-9][A-Za-z0-9._-]*\.jar$//' | sort | uniq -d)" +[[ -z "$duplicates" ]] || { echo "FAIL: duplicate connector artifacts:"; echo "$duplicates"; exit 1; } + +echo "==> [$IMAGE] connector classes resolve" +for class in \ + org.apache.hadoop.fs.s3a.S3AFileSystem \ + org.apache.hadoop.fs.azurebfs.oauth2.WorkloadIdentityTokenProvider \ + org.apache.hadoop.fs.gs.GoogleHadoopFileSystem; do + run "javap -classpath \"\$SPARK_HOME/jars/*\" '$class' >/dev/null" \ + || { echo "FAIL: connector class does not resolve: $class"; exit 1; } +done + echo "==> [$IMAGE] Python 3.11" python_version="$(run 'python3 --version 2>&1')" grep -q 'Python 3.11' <<<"$python_version" || { echo "FAIL: Python is not version 3.11: $python_version"; exit 1; } From a9d665c791d41d685dd611b540fabd885bc4d753 Mon Sep 17 00:00:00 2001 From: Gal Topper Date: Wed, 23 Sep 2026 18:38:27 +0700 Subject: [PATCH 5/5] Fix duplicate detection for Scala connector JARs --- docker/spark/scripts/validate-spark4.sh | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docker/spark/scripts/validate-spark4.sh b/docker/spark/scripts/validate-spark4.sh index cd58e3ee..863832aa 100644 --- a/docker/spark/scripts/validate-spark4.sh +++ b/docker/spark/scripts/validate-spark4.sh @@ -80,8 +80,8 @@ for jar in "${EXPECTED_CONNECTOR_JARS[@]}"; do done echo "==> [$IMAGE] no duplicate connector versions" -CONNECTOR_JAR_PREFIXES='^(hadoop-aws|hadoop-azure|hadoop-common|hadoop-gcp|aws-java-sdk-bundle|bundle|analyticsaccelerator-s3|wildfly-openssl|azure-storage|gcs-connector|jetty-util|jetty-util-ajax|spark-bigquery-with-dependencies)-' -duplicates="$(run 'ls "$SPARK_HOME/jars"' | grep -E "$CONNECTOR_JAR_PREFIXES" | sed -E 's/-[0-9][A-Za-z0-9._-]*\.jar$//' | sort | uniq -d)" +CONNECTOR_JAR_PREFIXES='^(hadoop-aws|hadoop-azure|hadoop-common|hadoop-gcp|aws-java-sdk-bundle|bundle|analyticsaccelerator-s3|wildfly-openssl|azure-storage|gcs-connector|jetty-util|jetty-util-ajax|spark-bigquery-with-dependencies(_[0-9]+\.[0-9]+)?)-' +duplicates="$(run 'ls "$SPARK_HOME/jars"' | grep -E "$CONNECTOR_JAR_PREFIXES" | sed -E 's/(_[0-9]+\.[0-9]+)?-[0-9][A-Za-z0-9._-]*\.jar$//' | sort | uniq -d)" [[ -z "$duplicates" ]] || { echo "FAIL: duplicate connector artifacts:"; echo "$duplicates"; exit 1; } echo "==> [$IMAGE] connector classes resolve"