diff --git a/.github/workflows/build-av.yml b/.github/workflows/build-av.yml index 86e2d6a..9e25234 100644 --- a/.github/workflows/build-av.yml +++ b/.github/workflows/build-av.yml @@ -79,7 +79,7 @@ jobs: sudo apt-get update sudo apt-get install -y ffmpeg uv venv - uv pip install pytest numpy twine 'opencv-python-headless>=4.12' 'Pillow>=11.3' dist/*.whl python-dist/*.whl + uv pip install pytest numpy twine 'Pillow>=11.3' dist/*.whl python-dist/*.whl - name: Generate baseline runtime fixtures run: uv run --no-sync python scripts/check_wheel_runtime.py generate .runtime-fixtures - name: Test on glibc 2.17 with Python 3.10 and 3.13 @@ -89,7 +89,7 @@ jobs: python="/opt/python/$abi/bin/python" "$python" -m venv "/tmp/$abi" runtime="/tmp/$abi/bin/python" - if [ "$abi" = cp310-cp310 ]; then "$runtime" -m pip install --only-binary=:all: numpy==1.26.4; fi + if [ "$abi" = cp310-cp310 ]; then "$runtime" -m pip install --only-binary=:all: numpy==2.0.2; fi "$runtime" -m pip install --only-binary=:all: /io/dist/*.whl /io/python-dist/*.whl env -u LD_LIBRARY_PATH "$runtime" /io/scripts/check_wheel_runtime.py check /io/.runtime-fixtures done diff --git a/.github/workflows/build-python.yml b/.github/workflows/build-python.yml index 3da6e25..41c6fb5 100644 --- a/.github/workflows/build-python.yml +++ b/.github/workflows/build-python.yml @@ -62,7 +62,7 @@ jobs: - name: Install tensorcodec without tensorcodec-av run: | python -m pip install pytest 'Pillow>=11.3' '${{ matrix.opencv }}' - python -m pip install --no-index --find-links dist "tensorcodec[images]" + python -m pip install --no-index --find-links dist tensorcodec python -c "import importlib.util as u; assert u.find_spec('tensorcodec_av') is None" - name: Test image codecs run: python -m pytest tests/test_images.py tests/test_image_encoders.py tests/test_av_optional.py tests/test_versions.py diff --git a/README.md b/README.md index 9688e12..656be41 100644 --- a/README.md +++ b/README.md @@ -26,7 +26,7 @@ Video, audio and image codecs with TorchCodec-style APIs and NumPy arrays. in a single native call, avoiding per-frame Python calls. Closing a decoder releases its FFmpeg resources without waiting for Python's cyclic GC. - **Lightweight installation.** Linux wheels are 10.9–11.1 MiB (v0.2.0), including - FFmpeg shared libraries. NumPy is the only required Python dependency. + FFmpeg shared libraries. NumPy and OpenCV (headless) are the only required Python dependencies. ## Quick start @@ -162,7 +162,7 @@ See the [compatibility contract](docs/compatibility.md) and These are `tensorcodec-av` wheels, installed automatically with `tensorcodec` (pure Python). Elsewhere only image codecs work; to build video/audio from source, install `tensorcodec-av==` ([external FFmpeg](docs/system_ffmpeg.md)). musl and - free-threaded CPython cannot be told apart by markers, so there use `pip install --no-deps tensorcodec numpy`. + free-threaded CPython cannot be told apart by markers, so there use `pip install --no-deps tensorcodec numpy opencv-python-headless`. - **Exact seeking:** scans packet timestamps when opening the decoder. Incorrect container keyframe flags can produce corrupt frames; repaired input or corrected frame mappings are needed in that case. diff --git a/av/Cargo.lock b/av/Cargo.lock index 571a4fc..ef67f69 100644 --- a/av/Cargo.lock +++ b/av/Cargo.lock @@ -449,7 +449,7 @@ checksum = "61c41af27dd6d1e27b1b16b489db798443478cef1f06a660c96db617ba5de3b1" [[package]] name = "tensorcodec-av" -version = "0.3.0" +version = "0.4.0" dependencies = [ "ffmpeg-sys-next", "libc", diff --git a/av/Cargo.toml b/av/Cargo.toml index a4da9b8..57a5e2f 100644 --- a/av/Cargo.toml +++ b/av/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "tensorcodec-av" -version = "0.3.0" +version = "0.4.0" edition = "2021" rust-version = "1.88" # ffmpeg-sys-next 9 builds with let chains license = "Apache-2.0" diff --git a/docs/images.md b/docs/images.md index 4b693be..f7d0981 100644 --- a/docs/images.md +++ b/docs/images.md @@ -1,11 +1,7 @@ # Image codecs Decode JPEG, PNG, WebP, GIF, AVIF and BMP into NumPy arrays, and encode grayscale or -RGB arrays as JPEG or PNG. Decoding returns CHW images or NCHW animations on CPU. - -```sh -uv pip install 'tensorcodec[images]' -``` +RGB arrays as JPEG or PNG (PNG also RGBA). Decoding returns CHW images or NCHW animations on CPU. ```python from tensorcodec.decoders import decode_image, decode_jpeg @@ -20,12 +16,11 @@ encoded = JpegEncoder(rgb).to_tensor(quality=90) # 1-D uint8 NumPy array ## Installation -The image backend uses OpenCV 4.12+ (tested with 4.12, 4.13 and 5.0); earlier -`opencv-python-headless` wheels lack the GIF and AVIF decoders. An existing compatible -`cv2` installation is sufficient; otherwise the `images` extra installs -`opencv-python-headless`. Use only one OpenCV wheel variant per environment. -NumPy is the only required dependency for the base package. Image dependencies -are separate from the base wheel size. Pillow is used only in tests. +`uv pip install tensorcodec` includes the image backend: `opencv-python-headless` 4.12+ +(tested with 4.12, 4.13 and 5.0), imported only when an image codec is first used. Earlier +wheels lack the GIF and AVIF decoders. Use only one OpenCV wheel variant per environment: +another variant such as `opencv-python` also provides `cv2` and conflicts with it. +Pillow is used only in tests. Image codecs do not need `tensorcodec-av`, so they also work on platforms without its wheels. @@ -53,7 +48,8 @@ Image codecs do not need `tensorcodec-av`, so they also work on platforms withou - Other formats depend on the installed OpenCV build (the Windows wheel has no AVIF decoder). Missing dependencies, unsupported codecs and decode failures raise; no alternate decoder is tried. -- Encoders accept nonempty CHW uint8 arrays with 1 or 3 channels. Both provide +- Encoders accept nonempty CHW uint8 arrays with 1 or 3 channels; `PngEncoder` also + accepts 4 (RGBA, alpha last, as `decode_image(..., mode="RGBA")` returns). Both provide `to_file`, `to_file_like` and `to_tensor`; JPEG quality is 1–100 (default 75), PNG compression level is 0–9 (default 6). Encoded bytes need not match TorchCodec. diff --git a/docs/package_size.md b/docs/package_size.md index 60adf23..f34813b 100644 --- a/docs/package_size.md +++ b/docs/package_size.md @@ -1,6 +1,6 @@ # Package size policy -TensorCodec keeps its NumPy-only Python dependency set and bundles a minimal +TensorCodec keeps its Python dependencies to NumPy and OpenCV and bundles a minimal FFmpeg/OpenSSL runtime in its default Linux `tensorcodec-av` wheels. Size limits prevent additions from silently increasing the distributed binary footprint. @@ -14,7 +14,7 @@ from silently increasing the distributed binary footprint. The sole policy file is [`packaging/size-policy.json`](../packaging/size-policy.json). The checker uses only Python's standard library and never extracts the archive. Unpacked size excludes filesystem allocation overhead. Both metrics exclude NumPy, -Python, package caches and other external dependencies; they are not total +OpenCV, Python, package caches and other external dependencies; they are not total installation sizes. Reports and the README use MiB (2^20 bytes). Check final, repaired wheels locally: diff --git a/docs/releasing.md b/docs/releasing.md index a0c1a91..3951c68 100644 --- a/docs/releasing.md +++ b/docs/releasing.md @@ -1,6 +1,6 @@ # Publishing TensorCodec -Release version: `0.3.0`. Each release publishes two PyPI projects at the same version: +Release version: `0.4.0`. Each release publishes two PyPI projects at the same version: `tensorcodec` (pure Python, built by hatchling) and `tensorcodec-av` (the FFmpeg extension in `av/`, built by maturin). `tensorcodec` pins `tensorcodec-av==` behind a platform marker; bump the version in `pyproject.toml` (twice), `av/Cargo.toml`, @@ -65,7 +65,7 @@ and source links are recorded in `av/licenses/README.md`. conda-forge dependency graph. - Release validation installs each repaired wheel with the `tensorcodec` wheel on glibc 2.17 with Python 3.10 and 3.13 and decodes video/audio without Torch, PyAV or a system FFmpeg. Python - 3.10 also checks the minimum NumPy line (1.26.4). Native + 3.10 also checks the minimum NumPy line (2.0.2). Native x86_64 and ARM64 runners also run the full pinned playback oracle comparison. - Release validation still tests the installed repaired wheel. The fixture CLI can be FFmpeg 6 or 7; fixtures explicitly remove auxiliary sentinel packets. diff --git a/docs/system_ffmpeg.md b/docs/system_ffmpeg.md index 0d4ae30..a023642 100644 --- a/docs/system_ffmpeg.md +++ b/docs/system_ffmpeg.md @@ -37,7 +37,7 @@ test -f "$FFMPEG_DIR/include/libavcodec/avcodec.h" test -f "$FFMPEG_DIR/lib/libavcodec.so.61" uv venv -uv pip install 'tensorcodec==0.3.0' 'tensorcodec-av==0.3.0' --no-binary tensorcodec-av +uv pip install 'tensorcodec==0.4.0' 'tensorcodec-av==0.4.0' --no-binary tensorcodec-av ``` Only `tensorcodec-av` is built from source; naming it also covers platforms diff --git a/pyproject.toml b/pyproject.toml index 8e98b4f..2ef0764 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "tensorcodec" -version = "0.3.0" +version = "0.4.0" description = "Video, audio and image codecs with TorchCodec-style APIs and NumPy arrays" readme = "README.md" license = "Apache-2.0" @@ -9,9 +9,11 @@ authors = [{ name = "Suhwan Choi", email = "milkclouds00@gmail.com" }] requires-python = ">=3.10" # Exact pin; the marker follows the tensorcodec-av wheel tags. No platform_release check (macOS 14): # pip's packaging 22-25 raises on Linux kernel releases that are not PEP 440 versions. +# opencv-python-headless 4.12+ requires NumPy 2. dependencies = [ - "numpy>=1.26", - "tensorcodec-av==0.3.0; platform_python_implementation == 'CPython' and ((sys_platform == 'linux' and (platform_machine == 'x86_64' or platform_machine == 'aarch64')) or (sys_platform == 'darwin' and platform_machine == 'arm64'))", + "numpy>=2", + "opencv-python-headless>=4.12", + "tensorcodec-av==0.4.0; platform_python_implementation == 'CPython' and ((sys_platform == 'linux' and (platform_machine == 'x86_64' or platform_machine == 'aarch64')) or (sys_platform == 'darwin' and platform_machine == 'arm64'))", ] classifiers = [ "Development Status :: 3 - Alpha", @@ -21,15 +23,12 @@ classifiers = [ "Topic :: Multimedia :: Video", ] -[project.optional-dependencies] -images = ["opencv-python-headless>=4.12"] - [project.urls] Repository = "https://github.com/MilkClouds/tensorcodec" Issues = "https://github.com/MilkClouds/tensorcodec/issues" [dependency-groups] -dev = ["pytest>=8", "ruff>=0.9", "maturin>=1.8,<2", "opencv-python-headless>=4.12", "Pillow>=11.3"] +dev = ["pytest>=8", "ruff>=0.9", "maturin>=1.8,<2", "Pillow>=11.3"] oracle = ["torch==2.14.1", "torchcodec==0.17.0"] [build-system] diff --git a/src/tensorcodec/__init__.py b/src/tensorcodec/__init__.py index 26f3a02..9b884ef 100644 --- a/src/tensorcodec/__init__.py +++ b/src/tensorcodec/__init__.py @@ -3,5 +3,5 @@ from tensorcodec import decoders, transforms from tensorcodec._frame import AudioSamples, Frame, FrameBatch -__version__ = "0.3.0" +__version__ = "0.4.0" __all__ = ["AudioSamples", "Frame", "FrameBatch", "decoders", "transforms"] diff --git a/src/tensorcodec/_opencv.py b/src/tensorcodec/_opencv.py index 5a63b15..6620b59 100644 --- a/src/tensorcodec/_opencv.py +++ b/src/tensorcodec/_opencv.py @@ -1,15 +1,10 @@ -"""Lazy access to the user's OpenCV installation.""" +"""Lazy access to OpenCV, which is imported only when an image codec is used.""" def opencv(): - try: - import cv2 - except ImportError as exc: - raise ImportError( - "Image codecs require OpenCV >= 4.12; install tensorcodec[images] " - "or use an existing compatible cv2 installation" - ) from exc + import cv2 + # 4.12: first PyPI wheels with GIF and AVIF decoders; 4.10/4.11 fail only those tests (docs/images.md). if tuple(int(part) for part in cv2.__version__.split(".")[:2]) < (4, 12): - raise ImportError("Image codecs require OpenCV >= 4.12") + raise ImportError(f"Image codecs require OpenCV >= 4.12, found {cv2.__version__}") return cv2 diff --git a/src/tensorcodec/decoders/_images.py b/src/tensorcodec/decoders/_images.py index 2bef3b7..88911f9 100644 --- a/src/tensorcodec/decoders/_images.py +++ b/src/tensorcodec/decoders/_images.py @@ -271,7 +271,7 @@ def decode_image(source, *, mode="RGB", output_dtype=np.uint8): Sources are paths, bytes or 1-D uint8 arrays. Modes: UNCHANGED, GRAY, GRAY_ALPHA, RGB, RGB_ALPHA (case-insensitive strings or ImageReadMode). output_dtype is uint8, uint16 or 'auto'; integer conversion scales the range. - BMP is detected too. Requires optional OpenCV >= 4.12. HEIC and animated PNG + BMP is detected too. Requires OpenCV >= 4.12. HEIC and animated PNG are unsupported. """ return _image(source, None, mode, output_dtype) diff --git a/src/tensorcodec/encoders/_images.py b/src/tensorcodec/encoders/_images.py index 98a96ae..e39b61c 100644 --- a/src/tensorcodec/encoders/_images.py +++ b/src/tensorcodec/encoders/_images.py @@ -8,18 +8,23 @@ class _ImageEncoder: + _channels = (1, 3) + def __init__(self, img): if not isinstance(img, np.ndarray): raise TypeError("img must be a NumPy array") - if img.dtype != np.uint8 or img.ndim != 3 or img.shape[0] not in (1, 3) or 0 in img.shape: - raise ValueError("img must be a nonempty CHW uint8 array with 1 or 3 channels") + if img.dtype != np.uint8 or img.ndim != 3 or img.shape[0] not in self._channels or 0 in img.shape: + channels = "/".join(map(str, self._channels)) + raise ValueError(f"{type(self).__name__} needs a nonempty CHW uint8 array with {channels} channels") self.img = img def _encode(self, extension, parameter, value, low, high): if isinstance(value, bool) or not isinstance(value, int) or not low <= value <= high: raise ValueError(f"encoding parameter must be an integer in [{low}, {high}]") cv = opencv() - pixels = self.img[0] if self.img.shape[0] == 1 else self.img.transpose(1, 2, 0)[..., ::-1] + channels = len(self.img) + # RGB(A) to OpenCV's BGR(A) + pixels = self.img[0] if channels == 1 else self.img.transpose(1, 2, 0)[..., [2, 1, 0, 3][:channels]] try: ok, encoded = cv.imencode(extension, np.ascontiguousarray(pixels), [getattr(cv, parameter), value]) except cv.error as exc: @@ -52,7 +57,9 @@ def to_tensor(self, *, quality=75): class PngEncoder(_ImageEncoder): - """Encode a CHW uint8 grayscale/RGB image on CPU.""" + """Encode a CHW uint8 grayscale/RGB/RGBA image on CPU.""" + + _channels = (1, 3, 4) def to_tensor(self, *, compression_level=6): """Return encoded PNG bytes as a one-dimensional uint8 NumPy array.""" diff --git a/tests/test_image_encoders.py b/tests/test_image_encoders.py index 46ea1f5..3e4581a 100644 --- a/tests/test_image_encoders.py +++ b/tests/test_image_encoders.py @@ -1,4 +1,4 @@ -"""Image encoder round trips, stream writes and optional dependency boundaries.""" +"""Image encoder round trips, stream writes and lazy OpenCV import.""" import subprocess import sys @@ -31,6 +31,17 @@ def test_encoder_outputs(encoder, channels, tmp_path): np.testing.assert_allclose(decode_image(result, mode="UNCHANGED").astype(int), pixels.astype(int), atol=2) +def test_png_rgba_round_trip(): + rng = np.random.default_rng(0) + pixels = rng.integers(0, 256, (4, 13, 17), dtype=np.uint8) + pixels[3, :2] = 0 # fully transparent rows keep their color + encoded = PngEncoder(pixels).to_tensor() + np.testing.assert_array_equal(np.array(Image.open(BytesIO(encoded.tobytes()))), pixels.transpose(1, 2, 0)) + for mode in ("UNCHANGED", "RGBA"): + np.testing.assert_array_equal(decode_image(encoded, mode=mode), pixels) + np.testing.assert_array_equal(decode_image(encoded), pixels[:3]) + + def test_png_noncontiguous_input_and_partial_writes(): pixels = np.arange(3 * 13 * 17, dtype=np.uint8).reshape(3, 13, 17)[:, ::-1, ::2] original = pixels.copy() @@ -64,7 +75,8 @@ def test_encoder_parameter_errors(encoder, key, values): "img", [ np.zeros((2, 2), np.uint8), - np.zeros((4, 2, 2), np.uint8), + np.zeros((2, 2, 2), np.uint8), + np.zeros((5, 2, 2), np.uint8), np.zeros((3, 0, 2), np.uint8), np.zeros((3, 2, 2), np.uint16), ], @@ -74,6 +86,11 @@ def test_encoder_input_errors(img): PngEncoder(img) +def test_jpeg_rejects_alpha(): + with pytest.raises(ValueError, match="JpegEncoder needs .* 1/3 channels"): + JpegEncoder(np.zeros((4, 2, 2), np.uint8)) + + def test_nonprogressing_writer(): class Writer: def write(self, data): @@ -83,7 +100,7 @@ def write(self, data): PngEncoder(np.zeros((3, 2, 2), np.uint8)).to_file_like(Writer()) -def test_cv2_is_lazy_and_optional(): +def test_cv2_is_lazy(): subprocess.run( [ sys.executable, @@ -98,8 +115,8 @@ def test_cv2_is_lazy_and_optional(): import numpy as np try: tensorcodec.encoders.PngEncoder(np.zeros((3, 2, 2), np.uint8)).to_tensor() -except ImportError as exc: - assert 'tensorcodec[images]' in str(exc) +except ImportError: + pass else: raise AssertionError('missing OpenCV was silently bypassed') """,