Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
33 changes: 33 additions & 0 deletions docker/spark/Dockerfile.spark4
Original file line number Diff line number Diff line change
@@ -0,0 +1,33 @@
# check=skip=InvalidDefaultArgInFrom
# Copyright 2026 Iguazio
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
# Spark 4 counterpart of ./Dockerfile.
# Spark 4.2.0, Scala 2.13, Temurin JDK 25, Hadoop 3.5.0 as bundled by the
# pinned upstream image.

# Required build argument; the Makefile supplies the pinned base reference.
ARG SPARK4_BASE_IMAGE
FROM ${SPARK4_BASE_IMAGE}

USER root

# Shared with the Spark 4 CUDA image.
COPY scripts/ce-customize-spark4.sh /tmp/ce-customize-spark4.sh
COPY scripts/jars-4.2.0.txt /tmp/jars-4.2.0.txt
RUN chmod +x /tmp/ce-customize-spark4.sh && \
/tmp/ce-customize-spark4.sh /tmp/jars-4.2.0.txt && \
rm -f /tmp/ce-customize-spark4.sh /tmp/jars-4.2.0.txt

USER spark
87 changes: 87 additions & 0 deletions docker/spark/Dockerfile.spark4.cuda
Original file line number Diff line number Diff line change
@@ -0,0 +1,87 @@
# check=skip=InvalidDefaultArgInFrom
# Copyright 2026 Iguazio
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
# CUDA counterpart of ./Dockerfile.spark4. The Spark distribution, the JDK,
# and the entrypoint are copied from the same digest-pinned base image the
# regular Spark 4 image is built from, so the two variants cannot drift and
# the build does not download Spark or the JDK a second time.

# Required build arguments; the Makefile supplies the pinned base references.
ARG SPARK4_BASE_IMAGE
ARG SPARK4_CUDA_BASE_IMAGE

FROM ${SPARK4_BASE_IMAGE} AS spark4

FROM ${SPARK4_CUDA_BASE_IMAGE}

# CUDA_VERSION and NV_CUDNN_VERSION are the CUDA base image's own environment.
# Do not declare an ARG of either name here: it would shadow the inherited
# value and let the labels describe a base this image was not built on.
LABEL com.iguazio.cuda-version="${CUDA_VERSION}" \
com.iguazio.cudnn-version="${NV_CUDNN_VERSION}"

ARG spark_uid=185

RUN groupadd --system --gid=${spark_uid} spark && \
useradd --system --uid=${spark_uid} --gid=spark -d /nonexistent spark

# Packages the upstream Spark image provides and the CUDA base does not.
# `locales` is one of them: the CUDA base generates only C and C.utf8, so
# en_US.UTF-8 has to be generated here for the locale ENV below to be valid.
RUN set -ex; \
export DEBIAN_FRONTEND=noninteractive; \
apt-get update; \
apt-get install -y --no-install-recommends \
gnupg2 bash tini libc6 libpam-modules \
krb5-user libnss3 procps net-tools gosu libnss-wrapper libjemalloc2 \
locales; \
locale-gen en_US.UTF-8; \
echo "auth required pam_wheel.so use_uid" >> /etc/pam.d/su; \
rm -rf /var/lib/apt/lists/*

# COPY carries no image environment, so the runtime settings from the Spark
# base are restated here. SPARK_TGZ_URL, SPARK_TGZ_ASC_URL, and GPG_KEY are
# build-provenance metadata and are intentionally omitted.
ENV LANG=en_US.UTF-8 \
LANGUAGE=en_US:en \
LC_ALL=en_US.UTF-8

COPY --from=spark4 /opt/java/openjdk /opt/java/openjdk
COPY --from=spark4 /opt/spark /opt/spark
COPY --from=spark4 /opt/decom.sh /opt/decom.sh
COPY --from=spark4 /opt/entrypoint.sh /opt/entrypoint.sh

# COPY creates the destination directory itself as root, while its contents
# keep the source image's ownership. Restore the Spark base's owner on the
# directory so the two variants match exactly.
RUN chown spark:spark /opt/spark

ENV JAVA_HOME=/opt/java/openjdk \
Comment thread
gtopper marked this conversation as resolved.
JAVA_VERSION=jdk-25.0.4+7 \
SPARK_HOME=/opt/spark
ENV PATH="${JAVA_HOME}/bin:${PATH}"

WORKDIR /opt/spark/work-dir

# Shared with the regular Spark 4 image.
COPY scripts/ce-customize-spark4.sh /tmp/ce-customize-spark4.sh
COPY scripts/jars-4.2.0.txt /tmp/jars-4.2.0.txt
RUN chmod +x /tmp/ce-customize-spark4.sh && \
/tmp/ce-customize-spark4.sh /tmp/jars-4.2.0.txt && \
rm -f /tmp/ce-customize-spark4.sh /tmp/jars-4.2.0.txt

USER spark

ENTRYPOINT [ "/opt/entrypoint.sh" ]
44 changes: 44 additions & 0 deletions docker/spark/Makefile
Original file line number Diff line number Diff line change
Expand Up @@ -27,3 +27,47 @@ validate-cuda:

.PHONY: validate-all
validate-all: validate validate-cuda

# Spark 4 variables and targets.
REGISTRY ?= gcr.io/iguazio
# Spark uses a multi-platform index; CUDA uses its linux/amd64 child manifest
# because the Spark 4 image pair is intentionally amd64-only.
SPARK4_BASE_IMAGE ?= spark@sha256:66e39dccde81909c23e5c56f4b465db6de2bdc38cf569aacc9e2380ee5005440
SPARK4_CUDA_BASE_IMAGE ?= nvidia/cuda@sha256:61f6c08f2b59036cb935e56d1e31a6b64e3ae2c7ddb86d33fa0b044c7917b719
MLRUN_CE_SPARK4_IMAGE_TAG ?= $(REGISTRY)/spark-app:4.2.0-scala2.13-java25-ubuntu-1
MLRUN_CE_SPARK4_CUDA_IMAGE_TAG ?= $(REGISTRY)/spark-app-cuda:4.2.0-scala2.13-java25-ubuntu-1

.PHONY: build-spark4
build-spark4:
docker build --platform=linux/amd64 \
--build-arg SPARK4_BASE_IMAGE="$(SPARK4_BASE_IMAGE)" \
-t "$(MLRUN_CE_SPARK4_IMAGE_TAG)" -f Dockerfile.spark4 .

.PHONY: build-spark4-cuda
build-spark4-cuda:
docker build --platform=linux/amd64 \
--build-arg SPARK4_BASE_IMAGE="$(SPARK4_BASE_IMAGE)" \
--build-arg SPARK4_CUDA_BASE_IMAGE="$(SPARK4_CUDA_BASE_IMAGE)" \
-t "$(MLRUN_CE_SPARK4_CUDA_IMAGE_TAG)" -f Dockerfile.spark4.cuda .

.PHONY: build-spark4-all
build-spark4-all: build-spark4 build-spark4-cuda

.PHONY: validate-spark4
validate-spark4:
bash scripts/validate-spark4.sh "$(MLRUN_CE_SPARK4_IMAGE_TAG)"

.PHONY: validate-spark4-cuda
validate-spark4-cuda:
bash scripts/validate-spark4.sh "$(MLRUN_CE_SPARK4_CUDA_IMAGE_TAG)" --cuda

.PHONY: validate-spark4-all
validate-spark4-all: validate-spark4 validate-spark4-cuda jar-parity-spark4 env-parity-spark4

.PHONY: jar-parity-spark4
jar-parity-spark4:
bash scripts/jar-parity.sh "$(MLRUN_CE_SPARK4_IMAGE_TAG)" "$(MLRUN_CE_SPARK4_CUDA_IMAGE_TAG)"

.PHONY: env-parity-spark4
env-parity-spark4:
bash scripts/env-parity.sh "$(MLRUN_CE_SPARK4_IMAGE_TAG)" "$(MLRUN_CE_SPARK4_CUDA_IMAGE_TAG)"
101 changes: 101 additions & 0 deletions docker/spark/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -49,3 +49,104 @@ docker inspect --format '{{index .RepoDigests 0}}' \

Do not republish the existing regular image. Record the CUDA image digest and
the CE source commit.

## Spark 4

The Spark 4 images target `linux/amd64` and use Spark 4.2.0, Scala 2.13,
Hadoop 3.5.0, Temurin 25.0.4+7, and Python 3.11:

- `spark-app:4.2.0-scala2.13-java25-ubuntu-1` uses the digest-pinned Spark
base configured by `SPARK4_BASE_IMAGE` in the Makefile.
- `spark-app-cuda:4.2.0-scala2.13-java25-ubuntu-1` copies Spark and Java from
that base into the digest-pinned `SPARK4_CUDA_BASE_IMAGE` (CUDA 12.8.1,
cuDNN 9.8.0.87-1).

Both images install the connector artifacts listed in
`scripts/jars-4.2.0.txt`: Hadoop 3.5.0 connectors for S3A, ABFS, and GCS,
their required AWS and Azure dependencies, and the Scala 2.13 BigQuery
connector. These restore the connector family shipped by the published Spark
3 image so Spark 4 can serve as a like-for-like platform image. Local
validation checks packaging, class resolution, duplicate versions, and
CPU/CUDA parity. Authenticated provider testing and BigQuery compatibility
with Spark 4.2 are deferred to the corresponding activation work.

`hadoop-gcp-3.5.0` provides
`org.apache.hadoop.fs.gs.GoogleHadoopFileSystem`, replacing the former
`com.google.cloud.hadoop.fs.gcs.GoogleHadoopFileSystem` class. It does not
provide a replacement `AbstractFileSystem` implementation. Consumers using
the old `fs.gs.impl` or `fs.AbstractFileSystem.gs.impl` configuration must
update or remove it.

`hadoop-aws-3.5.0` uses AWS SDK v2. Hadoop remaps several common SDK v1
credential-provider names, but not
`com.amazonaws.auth.DefaultAWSCredentialsProviderChain`, arbitrary
`com.amazonaws.*` providers, or custom SDK v1 implementations. Consumers
using those providers must migrate their configuration.

MLRun selects the CPU repository and tag through `MLRUN_SPARK_APP_IMAGE` and
`MLRUN_SPARK_APP_IMAGE_TAG`, and derives the CUDA repository by appending
`-cuda`. mlefi resolves both repositories and extracts the Spark version from
the tag with `^(.+?)-scala.*$`; for example,
`4.2.0-scala2.13-java25-ubuntu-1` yields `4.2.0`. These recipes do not change
the configured defaults.

### Build

```bash
make build-spark4
make build-spark4-cuda
make build-spark4-all
```

`REGISTRY` defaults to `gcr.io/iguazio`. The Makefile is the source of truth
for both pinned base-image digests.

### Validate

```bash
make validate-spark4-all
```

The validators check Spark, Scala, Java, Hadoop, Python, the connector
inventory, image metadata, and CUDA metadata. The aggregate target also checks
CPU/CUDA environment parity and compares every `$SPARK_HOME/jars` entry by
SHA-256. It does not authenticate to AWS, Azure, or GCP services.

### Publish (manual, JFrog)

Publication is manual. First check that neither
`spark-app:4.2.0-scala2.13-java25-ubuntu-1` nor
`spark-app-cuda:4.2.0-scala2.13-java25-ubuntu-1` already exists in
`mckinsey-ig4-next-gen-docker-local.jfrog.io`. Never overwrite an existing
immutable tag.

```bash
make REGISTRY=mckinsey-ig4-next-gen-docker-local.jfrog.io build-spark4-all
make REGISTRY=mckinsey-ig4-next-gen-docker-local.jfrog.io validate-spark4-all

docker login mckinsey-ig4-next-gen-docker-local.jfrog.io
docker push mckinsey-ig4-next-gen-docker-local.jfrog.io/spark-app:4.2.0-scala2.13-java25-ubuntu-1
docker push mckinsey-ig4-next-gen-docker-local.jfrog.io/spark-app-cuda:4.2.0-scala2.13-java25-ubuntu-1

docker inspect --format '{{index .RepoDigests 0}}' \
mckinsey-ig4-next-gen-docker-local.jfrog.io/spark-app:4.2.0-scala2.13-java25-ubuntu-1
docker inspect --format '{{index .RepoDigests 0}}' \
mckinsey-ig4-next-gen-docker-local.jfrog.io/spark-app-cuda:4.2.0-scala2.13-java25-ubuntu-1
```

Record both repository digests, then pull each image back by digest and
rerun validation against the digest references. A local image ID is not a
published digest.

### Jira evidence (ML-13080)

Attach to ML-13080:

- immutable tags and repository digests;
- pull-by-digest and image-inspection output;
- `spark-submit --version` and `java -version` output;
- CPU and CUDA JAR SHA-256 inventories and the parity result;
- the pinned base-image references from the Makefile;
- the source commit;
- the connector inventory and deferred provider-validation status documented
above.
50 changes: 50 additions & 0 deletions docker/spark/scripts/ce-customize-spark4.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,50 @@
#!/bin/bash
# Copyright 2026 Iguazio
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
# Spark 4 counterpart of ce-customize.sh. Installs Python 3.11, connector JARs
# from the supplied manifest, and a writable home directory for the spark user.
set -ex
export DEBIAN_FRONTEND=noninteractive

JAR_MANIFEST="${1:?usage: ce-customize-spark4.sh <jar-manifest>}"

apt-get update
apt-get install -y --no-install-recommends software-properties-common curl ca-certificates gnupg
add-apt-repository -y ppa:deadsnakes/ppa
apt-get update
apt-get install -y python3.11 python3.11-distutils git
rm -rf /var/lib/apt/lists/*

curl https://bootstrap.pypa.io/get-pip.py -o /tmp/get-pip.py
python3.11 /tmp/get-pip.py
rm -f /tmp/get-pip.py

while read -r line || [ -n "$line" ]; do
url="${line%%#*}"
url="$(echo "$url" | tr -d '[:space:]')"
[ -z "$url" ] && continue
case "$url" in
https://*) ;;
*) echo "bad manifest line: $line" >&2; exit 1 ;;
esac
curl -fsSL -o "/opt/spark/jars/$(basename "$url")" "$url"
done < "$JAR_MANIFEST"

update-alternatives --install /usr/bin/python3 python3 /usr/bin/python3.11 1
ln -sf /usr/bin/python3 /usr/bin/python

usermod -d /home/spark spark
mkdir -p /home/spark
chown spark:spark /home/spark
Loading
Loading