Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 3 additions & 3 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -35,7 +35,7 @@ jobs:
run: |
python -m venv .venv
.venv/bin/python -m pip install --upgrade pip
.venv/bin/pip install numpy pytest ruff 'maturin>=1.8,<2'
.venv/bin/pip install numpy pytest ruff 'maturin>=1.8,<2' 'opencv-python-headless>=4.13,<5' 'Pillow>=11.3'
echo "$GITHUB_WORKSPACE/.venv/bin" >> "$GITHUB_PATH"
echo "VIRTUAL_ENV=$GITHUB_WORKSPACE/.venv" >> "$GITHUB_ENV"
- name: Configure prebuilt FFmpeg
Expand All @@ -51,8 +51,8 @@ jobs:
"$prefix/bin/ffmpeg" -version
- name: Static checks
run: |
ruff check src/tensorcodec tests scripts
ruff format --check src/tensorcodec tests scripts
ruff check src/tensorcodec tests scripts benchmarks/image_codecs.py
ruff format --check src/tensorcodec tests scripts benchmarks/image_codecs.py
cargo fmt --manifest-path native/Cargo.toml --check
cargo clippy --manifest-path native/Cargo.toml --locked -- -D warnings
- name: Build extension against prebuilt FFmpeg
Expand Down
2 changes: 1 addition & 1 deletion .github/workflows/macos-wheels.yml
Original file line number Diff line number Diff line change
Expand Up @@ -33,7 +33,7 @@ jobs:
run: |
brew install nasm pkg-config coreutils meson ninja
uv venv
uv pip install numpy pytest 'maturin>=1.8,<2' delocate twine
uv pip install numpy pytest 'opencv-python-headless>=4.13,<5' 'Pillow>=11.3' 'maturin>=1.8,<2' delocate twine
echo "$PWD/.venv/bin" >> "$GITHUB_PATH"
echo "LIBCLANG_PATH=$(xcode-select -p)/Toolchains/XcodeDefault.xctoolchain/usr/lib" >> "$GITHUB_ENV"
echo "$HOME/.pixi/envs/ffmpeg/bin" >> "$GITHUB_PATH"
Expand Down
2 changes: 1 addition & 1 deletion .github/workflows/publish.yml
Original file line number Diff line number Diff line change
Expand Up @@ -87,7 +87,7 @@ jobs:
sudo apt-get update
sudo apt-get install -y ffmpeg
uv venv
uv pip install pytest numpy twine dist/*.whl
uv pip install pytest numpy twine 'opencv-python-headless>=4.13,<5' 'Pillow>=11.3' dist/*.whl
- name: Generate baseline runtime fixtures
run: uv run --no-sync python scripts/check_wheel_runtime.py generate .runtime-fixtures
- name: Test on glibc 2.17 with Python 3.10 and 3.13
Expand Down
3 changes: 2 additions & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@
# TensorCodec

CPU video/audio decoding with TorchCodec-style APIs and NumPy output.
Optional image decoding and JPEG/PNG encoding reuse OpenCV.

<p align="center">
<a href="https://github.com/MilkClouds/tensorcodec/actions/workflows/ci.yml"><img src="https://github.com/MilkClouds/tensorcodec/actions/workflows/ci.yml/badge.svg?branch=main" alt="CI"></a>
Expand All @@ -14,7 +15,7 @@ CPU video/audio decoding with TorchCodec-style APIs and NumPy output.
<a href="LICENSE"><img src="https://img.shields.io/badge/License-Apache--2.0-blue" alt="License: Apache-2.0"></a>
</p>

[Quick start](#quick-start) · [Features](#features) · [Package size](#package-size) · [Compatibility](docs/compatibility.md)
[Quick start](#quick-start) · [Features](#features) · [Package size](#package-size) · [Compatibility](docs/compatibility.md) · [Image codecs](docs/images.md)

</div>

Expand Down
65 changes: 65 additions & 0 deletions benchmarks/image_codecs.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,65 @@
"""Compare RGB decoding on a directory of JPEG/PNG/WebP files; run inside a uv venv."""

import argparse
import json
import os
import statistics
import time
from pathlib import Path

import cv2
import numpy as np
import torch
import torchcodec
from torchcodec.decoders import decode_image as reference

from tensorcodec.decoders import decode_image


def latency(fn):
for _ in range(3):
fn()
start = time.perf_counter()
fn()
count = max(2, min(50, round(0.025 / (time.perf_counter() - start))))
times = []
for _ in range(5):
start = time.perf_counter()
for _ in range(count):
fn()
times.append((time.perf_counter() - start) * 1000 / count)
return statistics.median(times)


def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("corpus", type=Path)
args = parser.parse_args()
if hasattr(os, "sched_getaffinity"):
os.sched_setaffinity(0, {min(os.sched_getaffinity(0))})
cv2.setNumThreads(1)
torch.set_num_threads(1)
print(json.dumps({"opencv": cv2.__version__, "torchcodec": torchcodec.__version__}))
for path in sorted(args.corpus.iterdir()):
if not path.is_file():
continue
data = path.read_bytes()
if not (data.startswith((b"\xff\xd8\xff", b"\x89PNG")) or data[8:12] == b"WEBP"):
continue
actual, expected = decode_image(data), reference(data).numpy()
np.testing.assert_array_equal(actual, expected, err_msg=path.name)
print(
json.dumps(
{
"file": path.name,
"max_error": 0,
"tensorcodec_ms": latency(lambda data=data: decode_image(data)),
"torchcodec_ms": latency(lambda data=data: reference(data)),
}
),
flush=True,
)


if __name__ == "__main__":
main()
79 changes: 79 additions & 0 deletions docs/images.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,79 @@
# Optional CPU image codecs

Image decoding and JPEG/PNG encoding use the user's OpenCV installation through
Python. NumPy remains the only required Python dependency; importing TensorCodec
does not import OpenCV or Pillow. No native image libraries or build steps are
added to TensorCodec. Pillow is used only to generate independent test inputs.

Use an existing `cv2 >= 4.13` installation, or install one explicitly:

```sh
uv pip install 'tensorcodec[images]'
```

The extra selects `opencv-python-headless`. Do not install multiple OpenCV wheel
variants in the same environment. OpenCV adds its own wheel size and dependencies;
it is not part of TensorCodec's advertised base wheel size.

```python
from tensorcodec.decoders import decode_image, decode_jpeg
from tensorcodec.encoders import JpegEncoder, PngEncoder

rgb = decode_image('input.webp') # uint8 CHW, RGB
frames = decode_image('animation.gif') # NCHW for multiple frames
batch = decode_jpeg(['one.jpg', 'two.jpg']) # list, possibly different sizes
PngEncoder(rgb).to_file('output.png', compression_level=6)
encoded = JpegEncoder(rgb).to_tensor(quality=90) # 1-D uint8 NumPy array
```

## Contract and limits

- `decode_image`, `decode_jpeg`, `decode_png`, `decode_webp`, `decode_gif`,
`decode_avif` accept paths, bytes, bytearray or 1-D uint8 arrays.
- `mode` accepts case-insensitive `UNCHANGED`, `GRAY`, `GRAY_ALPHA`, `RGB`
(default), `RGB_ALPHA`/`RGBA`, or `ImageReadMode` values.
- `output_dtype` accepts uint8 (default), uint16 or `"auto"`. PNG preserves native
8/16-bit precision with `"auto"`; explicit conversions scale the integer range.
- Multiple frames return NCHW; still images return CHW. Animated WebP keeps NCHW
even with one frame. Frame timings and loop counts are not returned.
- JPEG/PNG/WebP EXIF orientation and AVIF primary-item rotation/mirror are applied.
AVIF track-specific transforms are outside this adapter's contract.
- APNG, high-bit-depth AVIF, CMYK JPEG `UNCHANGED`, GPU decoding and nondefault
AVIF `num_threads` raise explicit errors. OpenCV has no per-call AVIF thread
setting; the adapter never modifies global OpenCV thread settings.
- AVIF color conversion follows OpenCV. Dropping alpha preserves straight RGB;
TorchCodec 0.17.0 premultiplies AVIF RGB in that case. Pixel identity with
TorchCodec is not promised across formats, builds or codec versions.
- HEIC is unsupported. There is no separate libheif or other decoder fallback.
- Other formats depend on the installed OpenCV build. Missing dependencies,
unsupported codecs and decode failures raise; no alternate decoder is tried.
- Encoders accept nonempty CHW uint8 arrays with 1 or 3 channels. Both provide
`to_file`, `to_file_like` and `to_tensor`; JPEG quality is 1–100 (default 75),
PNG compression level is 0–9 (default 6). Encoded bytes need not match TorchCodec.

## Validation and performance

Tests cover known PNG samples, independent Pillow decoding of encoder outputs,
orientation, animations, malformed input and TorchCodec 0.17.0 comparisons.
Pillow is a test dependency, not a runtime backend.

On one Linux CPU, the adapter matched TorchCodec exactly for RGB output on 54
JPEG/PNG/WebP inputs: photograph, graphics and seeded noise at 224 square,
640×480 and 1920×1080. OpenCV was 4.13.0.92. Timings used one pinned CPU,
one OpenCV/Torch thread, three warmups and the median of five batches.

| Encoding | Adapter / TorchCodec latency, geometric mean |
| --- | ---: |
| JPEG 4:2:0 | 1.05 |
| JPEG 4:4:4 | 1.03 |
| Progressive JPEG | 1.02 |
| PNG | 1.28 |
| Lossy WebP | 1.07 |
| Lossless WebP | 1.11 |

These corpus-specific results favor simplicity over specialized native backends;
OpenCV is not universally fastest. Encoding speed has not been benchmarked.
With the development and oracle dependencies installed, run
`python benchmarks/image_codecs.py CORPUS_DIRECTORY` inside a uv virtual
environment to measure in-memory decoding and exact RGB agreement. The script
prints versions and per-file results; disk reads are outside timed sections.
5 changes: 4 additions & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -17,12 +17,15 @@ classifiers = [
"Topic :: Multimedia :: Video",
]

[project.optional-dependencies]
images = ["opencv-python-headless>=4.13,<5"]

[project.urls]
Repository = "https://github.com/MilkClouds/tensorcodec"
Issues = "https://github.com/MilkClouds/tensorcodec/issues"

[dependency-groups]
dev = ["pytest>=8", "ruff>=0.9", "maturin>=1.8,<2"]
dev = ["pytest>=8", "ruff>=0.9", "maturin>=1.8,<2", "opencv-python-headless>=4.13,<5", "Pillow>=11.3"]
oracle = ["torch==2.14.1", "torchcodec==0.17.0"]

[build-system]
Expand Down
14 changes: 14 additions & 0 deletions src/tensorcodec/_opencv.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,14 @@
"""Lazy access to the user's OpenCV installation."""


def opencv():
try:
import cv2
except ImportError as exc:
raise ImportError(
"Image codecs require OpenCV >= 4.13; install tensorcodec[images] "
"or use an existing compatible cv2 installation"
) from exc
if tuple(int(part) for part in cv2.__version__.split(".")[:2]) < (4, 13):
raise ImportError("Image codecs require OpenCV >= 4.13")
return cv2
24 changes: 23 additions & 1 deletion src/tensorcodec/decoders/__init__.py
Original file line number Diff line number Diff line change
@@ -1,4 +1,26 @@
from tensorcodec._metadata import AudioStreamMetadata, VideoStreamMetadata
from tensorcodec.decoders._decoder import AudioDecoder, CpuFallbackStatus, VideoDecoder
from tensorcodec.decoders._images import (
ImageReadMode,
decode_avif,
decode_gif,
decode_image,
decode_jpeg,
decode_png,
decode_webp,
)

__all__ = ["AudioDecoder", "AudioStreamMetadata", "CpuFallbackStatus", "VideoDecoder", "VideoStreamMetadata"]
__all__ = [
"AudioDecoder",
"AudioStreamMetadata",
"CpuFallbackStatus",
"ImageReadMode",
"VideoDecoder",
"VideoStreamMetadata",
"decode_avif",
"decode_gif",
"decode_image",
"decode_jpeg",
"decode_png",
"decode_webp",
]
Loading
Loading