Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 3 additions & 9 deletions .github/workflows/check-python-code.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -61,15 +61,9 @@ jobs:
with:
python-version: ${{ matrix.python }}

# PySpark 3.4 (the declared minimum) does not support Java 21, which is
# the default JDK on ubuntu-latest runners. Pin to Java 17 for the
# lowest-direct cell so the resolved pyspark==3.4.0 can actually start.
- name: Set up JDK 17
if: matrix.resolution == 'lowest-direct'
uses: actions/setup-java@03ad4de0992f5dab5e18fcb136590ce7c4a0ac95 # v5.6.0
with:
distribution: temurin
java-version: '17'
# No setup-java: the test suite runs on the jdk4py JDK from the dev
# group. Its floor (JDK 17) is what lowest-direct resolves, matching
# pyspark 3.4, which does not support the runners' default Java 21.

# UV_RESOLUTION=lowest-direct makes `uv sync` re-resolve every direct
# dependency to the lowest version permitted by pyproject.toml. This
Expand Down
86 changes: 82 additions & 4 deletions packages/overture-schema-pyspark/tests/conftest.py
Original file line number Diff line number Diff line change
@@ -1,18 +1,96 @@
"""Shared pytest fixtures for overture-schema-pyspark tests."""

import atexit
import os
import shutil
import socket
import sys
import tempfile
from collections.abc import Callable
from pathlib import Path
from typing import Any

import pyspark
import pytest
from pyspark.sql import SparkSession

# Ensure PySpark workers use the same Python as the driver to avoid
# version mismatch errors when a different system Python is on PATH.
os.environ.setdefault("PYSPARK_PYTHON", sys.executable)
os.environ.setdefault("PYSPARK_DRIVER_PYTHON", sys.executable)
# Tests must be hermetic against ambient Spark/JVM configuration: developers
# commonly have a system-wide Spark install or JVM flags exported for other
# projects, and any of these leaking into the test JVM changes -- or breaks --
# every session. Networking overrides such as SPARK_LOCAL_IP are deliberately
# left alone: those are set to make local sessions *work* (e.g. macOS/VPN
# hostname resolution).
for _var in (
"PYSPARK_SUBMIT_ARGS", # injects launcher/JVM flags, jars, even --master
"SPARK_CONF_DIR", # points at another install's spark-defaults.conf
"HADOOP_CONF_DIR", # redirects filesystem/cluster defaults
"YARN_CONF_DIR",
"JAVA_TOOL_OPTIONS", # picked up unconditionally by every JVM
"_JAVA_OPTIONS",
"JDK_JAVA_OPTIONS",
):
os.environ.pop(_var, None)
# Pin worker and driver Python to this interpreter (overriding, not
# setdefault): an exported PYSPARK_PYTHON pointing at a non-venv Python
# fails with pickle/version-mismatch errors inside executors.
os.environ["PYSPARK_PYTHON"] = sys.executable
os.environ["PYSPARK_DRIVER_PYTHON"] = sys.executable
# Pin Spark to the venv's bundled distribution: an ambient SPARK_HOME
# pointing at a system-wide Spark install would launch that JVM under this
# venv's PySpark client, and mismatched versions fail at session setup with
# opaque py4j "'JavaPackage' object is not callable" errors.
os.environ["SPARK_HOME"] = os.path.dirname(pyspark.__file__)


def _shimmed_java_home(java_home: Path) -> Path:
"""Mirror *java_home*, replacing bin/java with a flag-filtering wrapper.

jdk4py ships a jlink-stripped runtime without jdk.incubator.vector
(upstream declined to bundle it for size: atoti/jdk4py#95), while
Spark 4.x passes --add-modules=jdk.incubator.vector to every JVM it
launches -- and a JVM refuses to boot when an --add-modules target is
missing. The module only enables vectorized BLAS, which Spark treats
as optional at runtime, so the wrapper drops that one flag and execs
the real java with everything else intact.
"""
shim_home = Path(tempfile.mkdtemp(prefix="overture-pyspark-test-jdk-"))
atexit.register(shutil.rmtree, shim_home, ignore_errors=True)
for entry in java_home.iterdir():
if entry.name != "bin":
(shim_home / entry.name).symlink_to(entry)
bin_dir = shim_home / "bin"
bin_dir.mkdir()
for entry in (java_home / "bin").iterdir():
if entry.name != "java":
(bin_dir / entry.name).symlink_to(entry)
real_java = java_home / "bin" / "java"
shim_java = bin_dir / "java"
shim_java.write_text(
"#!/bin/sh\n"
"# Drop --add-modules=jdk.incubator.vector; see conftest.py.\n"
"for arg do\n"
" shift\n"
' [ "$arg" = "--add-modules=jdk.incubator.vector" ] && continue\n'
' set -- "$@" "$arg"\n'
"done\n"
f'exec "{real_java}" "$@"\n'
)
shim_java.chmod(0o755)
return shim_home


# Pin the JVM to the venv's bundled JDK (jdk4py, a dev dependency on the
# platforms that have wheels for it). On platforms without a wheel -- or
# when the dev group isn't installed -- fall back to the machine's Java.
# The shim wrapper is a POSIX shell script, so non-POSIX platforms also
# fall back.
try:
import jdk4py
except ImportError:
pass
else:
if os.name == "posix":
os.environ["JAVA_HOME"] = str(_shimmed_java_home(jdk4py.JAVA_HOME))


def pytest_configure(config: pytest.Config) -> None:
Expand Down
8 changes: 8 additions & 0 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -50,6 +50,14 @@ warn_unused_configs = true

[dependency-groups]
dev = [
# Bundled JDK for the pyspark test suite, so tests never depend on the
# machine's Java. <22 keeps every resolution inside Spark's supported
# range: the floor (17.0.9.2 = JDK 17) is what lowest-direct resolves
# for pyspark 3.4, and the lock picks the maintained 21.x line for
# pyspark 4.x. jdk4py ships wheels only and the conftest shim that
# adapts it for Spark is POSIX shell, so gate it to darwin/linux;
# elsewhere conftest falls back to the system Java.
"jdk4py>=17.0.9.2,<22; sys_platform == 'darwin' or sys_platform == 'linux'",
"mypy>=1.17.0",
"pdoc>=15.0.4",
"pydocstyle>=6.3.0",
Expand Down
17 changes: 15 additions & 2 deletions uv.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

Loading