diff --git a/.github/workflows/build-check.yml b/.github/workflows/build-check.yml index af75d6111b85..1a4143ca4729 100644 --- a/.github/workflows/build-check.yml +++ b/.github/workflows/build-check.yml @@ -64,7 +64,10 @@ jobs: --prefix=/usr \ -Dplatforms=x11,wayland \ -Dgallium-drivers=freedreno,zink,virgl,llvmpipe \ - -Dgallium-va=disabled \ + -Dgallium-va=enabled \ + -Dtermux-va-bridge=enabled \ + -Dtermux-va-wayland-shim=enabled \ + -Dvideo-codecs=all \ -Dgallium-mediafoundation=disabled \ -Dvulkan-drivers=freedreno \ -Dvulkan-layers= \ @@ -165,7 +168,10 @@ jobs: --prefix=/usr \ -Dplatforms=x11,wayland \ -Dgallium-drivers=freedreno,zink,virgl,llvmpipe \ - -Dgallium-va=disabled \ + -Dgallium-va=enabled \ + -Dtermux-va-bridge=enabled \ + -Dtermux-va-wayland-shim=enabled \ + -Dvideo-codecs=all \ -Dgallium-mediafoundation=disabled \ -Dvulkan-drivers=freedreno \ -Dvulkan-layers= \ @@ -271,7 +277,10 @@ jobs: --prefix=/usr \ -Dplatforms=x11,wayland \ -Dgallium-drivers=freedreno,zink,virgl,llvmpipe \ - -Dgallium-va=disabled \ + -Dgallium-va=enabled \ + -Dtermux-va-bridge=enabled \ + -Dtermux-va-wayland-shim=enabled \ + -Dvideo-codecs=all \ -Dgallium-mediafoundation=disabled \ -Dvulkan-drivers=freedreno \ -Dvulkan-layers= \ @@ -375,7 +384,10 @@ jobs: --prefix=/usr \ -Dplatforms=x11,wayland \ -Dgallium-drivers=freedreno,zink,virgl,llvmpipe \ - -Dgallium-va=disabled \ + -Dgallium-va=enabled \ + -Dtermux-va-bridge=enabled \ + -Dtermux-va-wayland-shim=enabled \ + -Dvideo-codecs=all \ -Dgallium-mediafoundation=disabled \ -Dvulkan-drivers=freedreno \ -Dvulkan-layers= \ @@ -494,7 +506,10 @@ jobs: --prefix=/usr \ -Dplatforms=x11,wayland \ -Dgallium-drivers=freedreno,zink,virgl,llvmpipe \ - -Dgallium-va=disabled \ + -Dgallium-va=enabled \ + -Dtermux-va-bridge=enabled \ + -Dtermux-va-wayland-shim=enabled \ + -Dvideo-codecs=all \ -Dgallium-mediafoundation=disabled \ -Dvulkan-drivers=freedreno \ -Dvulkan-layers= \ @@ -598,7 +613,10 @@ jobs: --prefix=/usr \ -Dplatforms=x11,wayland \ -Dgallium-drivers=freedreno,zink,virgl,llvmpipe \ - -Dgallium-va=disabled \ + -Dgallium-va=enabled \ + -Dtermux-va-bridge=enabled \ + -Dtermux-va-wayland-shim=enabled \ + -Dvideo-codecs=all \ -Dgallium-mediafoundation=disabled \ -Dvulkan-drivers=freedreno \ -Dvulkan-layers= \ @@ -702,7 +720,10 @@ jobs: --prefix=/usr \ -Dplatforms=x11,wayland \ -Dgallium-drivers=freedreno,zink,virgl,llvmpipe \ - -Dgallium-va=disabled \ + -Dgallium-va=enabled \ + -Dtermux-va-bridge=enabled \ + -Dtermux-va-wayland-shim=enabled \ + -Dvideo-codecs=all \ -Dgallium-mediafoundation=disabled \ -Dvulkan-drivers=freedreno \ -Dvulkan-layers= \ @@ -775,6 +796,7 @@ jobs: python3-Mako python3-packaging python3-yaml python3-ply \ libffi-devel \ libzstd-devel \ + libva-devel \ glslang - name: Cache ccache @@ -818,7 +840,10 @@ jobs: --prefix=/usr \ -Dplatforms=x11,wayland \ -Dgallium-drivers=freedreno,zink,virgl,llvmpipe \ - -Dgallium-va=disabled \ + -Dgallium-va=enabled \ + -Dtermux-va-bridge=enabled \ + -Dtermux-va-wayland-shim=enabled \ + -Dvideo-codecs=all \ -Dgallium-mediafoundation=disabled \ -Dvulkan-drivers=freedreno \ -Dvulkan-layers= \ @@ -923,7 +948,10 @@ jobs: --prefix=/usr \ -Dplatforms=x11,wayland \ -Dgallium-drivers=freedreno,zink,virgl,llvmpipe \ - -Dgallium-va=disabled \ + -Dgallium-va=enabled \ + -Dtermux-va-bridge=enabled \ + -Dtermux-va-wayland-shim=enabled \ + -Dvideo-codecs=all \ -Dgallium-mediafoundation=disabled \ -Dvulkan-drivers=freedreno \ -Dvulkan-layers= \ diff --git a/docs/envvars.rst b/docs/envvars.rst index 1ebe2c361cef..fe28b64c84f7 100644 --- a/docs/envvars.rst +++ b/docs/envvars.rst @@ -1404,6 +1404,108 @@ VA-API environment variables enable MPEG4 for VA-API, disabled by default. +termux-va bridge environment variables +-------------------------------------- + +The termux-va bridge forwards VA-API video decoding over a Unix socket to +the termux-va daemon running in Termux (Android MediaCodec hardware +decode). See :doc:`termux-va`. + +.. envvar:: TERMUX_VA_BRIDGE + + ``1``/``true`` forces the bridge on, ``0``/``false`` forces it off. + When unset, the bridge activates automatically if ``TERMUX_VA_SOCKET`` + or ``TERMUX_VA_SOCKET_DIR`` is set, or if the default endpoint exists + as a socket. + +.. envvar:: TERMUX_VA_SOCKET + + Full path of the daemon's Unix socket file, overriding the default + ``/tmp/termux-va/termux-va.sock`` (the container-side view of the + Termux ``$TMPDIR/termux-va/`` shared-tmp directory). + +.. envvar:: TERMUX_VA_SOCKET_DIR + + Directory containing the socket; ``termux-va.sock`` is appended. + Takes effect when ``TERMUX_VA_SOCKET`` is unset. The same two + variables are understood by the daemon itself, so one setting covers + both ends. + +.. envvar:: TERMUX_VA_GPU_BACKEND + + selects how the bridge creates the underlying screen that hosts the + decode surfaces: + + - ``auto`` (default): try the stock loader and fall back to llvmpipe. The KGSL alias is not attempted automatically. + - ``kgsl``: force the KGSL Freedreno alias. GPU submission uses + ``/dev/kgsl-3d0`` while the handed fd remains the control/identity fd. + EGL and GLX loader selection follows this value, so + ``MESA_LOADER_DRIVER_OVERRIDE=kgsl`` and ``FD_FORCE_KGSL=1`` are not + required separately. + - ``drm``: use stock loader selection only. + - ``sw``: use llvmpipe only; no GPU is needed for the CPU frame-copy paths. + +.. envvar:: TERMUX_VA_DISABLE_AVC + + set to ``1``, ``true`` or ``on`` to hide the H.264/AVC profiles from + libva. ``TERMUX_VA_DISABLE_H264`` is accepted as a compatibility alias. + +.. envvar:: TERMUX_VA_DISABLE_HEVC + + set to ``1``, ``true`` or ``on`` to hide the HEVC Main profile from libva. + +.. envvar:: TERMUX_VA_DISABLE_VP9 + + set to ``1``, ``true`` or ``on`` to hide the VP9 Profile 0 profile from + libva. + +.. envvar:: TERMUX_VA_DISABLE_AV1 + + set to ``1``, ``true`` or ``on`` to hide the AV1 Main profile from libva. + This is useful when Android exposes only a software MediaCodec component + for AV1, so applications can use their native software decoder instead. + +.. envvar:: DMD_WANT_SHM + + set to ``0`` to disable the memfd shared-memory frame transport + (zero-copy) and always receive frames inline on the socket. The + ``DMD_`` prefix is kept for compatibility with the upstream protocol + tooling. + +.. envvar:: TERMUX_VA_PIPELINE_DEPTH + + sets the bridge's normal pending-picture depth to a value from 2 to 32 + (default 6). + When shared-memory transport is enabled, the value is clamped to the + daemon's ``SHM_SLOTS`` limit. + +.. envvar:: DMD_VA_CPU_COPY + + controls how staged frames are copied into bridge surfaces. On the KGSL + backend CPU-mapped writes are enabled by default to make the cache handoff + to a separate Vulkan/KGSL importer explicit. Set to ``0``, ``false`` or + ``off`` to retain the asynchronous Gallium ``texture_subdata`` path; set + to any other non-empty value to force CPU copies. + +.. envvar:: DMD_VA_CONTIGUOUS_DMABUF + + controls the NV12 surface layout exported to consumers. Set to ``1``, + ``true`` or ``on`` to place both planes in one dma-buf, or to ``0`` (or any + other value) to retain separate plane objects. When unset, the bridge + automatically selects the single-object layout for a KGSL-only container + that has no DRM render node; this is required by Chromium's current native + pixmap importer. ``TERMUX_VA_CONTIGUOUS_DMABUF`` is an alias. + +.. envvar:: DMD_VA_LOG + + set to ``1`` to enable the bridge's daemon-client logging on stderr. + +.. envvar:: LIBVA_DRIVER_NAME + + set to ``termuxva`` to make libva load the bridge through the + ``termuxva_drv_video.so`` megadriver symlink (recommended; automatic + driver discovery does not know the bridge). + VC4 driver environment variables -------------------------------- diff --git a/docs/index.rst b/docs/index.rst index 7c484f7880ab..4d24a52ca653 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -121,6 +121,7 @@ Linux, FreeBSD, and other operating systems. isaspec rusticl android + termux-va macos Linux Kernel Drivers diff --git a/docs/termux-va.rst b/docs/termux-va.rst new file mode 100644 index 000000000000..6a7e12321acd --- /dev/null +++ b/docs/termux-va.rst @@ -0,0 +1,139 @@ +termux-va bridge +================ + +The termux-va bridge forwards VA-API video decoding from this Mesa build +(the container side) to the `termux-va` daemon running in Termux on the +Android host, which decodes with the Android MediaCodec API in hardware +and returns NV12 frames. Applications inside a Linux container that +shares Termux's tmp directory (``proot-distro ... --shared-tmp``) get +hardware decoding through the standard VA-API without any modification. + +The porting model follows anland-termux: a Termux daemon, a Unix socket +placed in the shared tmp directory, and a bridge on the container side. +The daemon lives in the `termux-va` repository; the wire protocol is +byte-compatible with droidspaces-media-decode protocol v3. + +Supported codecs: H.264 (Constrained Baseline / Main / High), HEVC Main, VP9 Profile 0, and AV1 Main, outputting NV12 progressive frames. Profiles are advertised to libva through the underlying screen; encode and other codecs are not provided. + +Building +-------- + +Build with ``-Dgallium-va=enabled -Dtermux-va-bridge=enabled`` and at +least one of ``h264dec``, ``h265dec``, ``vp9dec`` in ``video-codecs`` +(for example ``-Dvideo-codecs=all``). The megadriver is additionally +exposed as ``termuxva_drv_video.so`` so libva can select it with +``LIBVA_DRIVER_NAME=termuxva``. Native Wayland Chromium support also +requires ``-Dtermux-va-wayland-shim=enabled``; this installs the +process-local DRM compatibility shim described below. + +Activation +---------- + +The bridge is runtime-gated; a Mesa build with the bridge behaves exactly +like an unmodified one until activation: + +- ``TERMUX_VA_BRIDGE=1`` forces the bridge on, ``0`` forces it off. +- Unset: the bridge activates when ``TERMUX_VA_SOCKET`` / + ``TERMUX_VA_SOCKET_DIR`` is set, or when the default endpoint exists as + a socket. + +When the bridge is active but the daemon is unreachable, driver init +fails cleanly and applications fall back to software decoding. + +The bridge can selectively hide hardware decode profiles from libva. Set +``TERMUX_VA_DISABLE_AVC=1`` (or the compatibility alias +``TERMUX_VA_DISABLE_H264=1``) to disable H.264/AVC, or set +``TERMUX_VA_DISABLE_HEVC=1``, ``TERMUX_VA_DISABLE_VP9=1``, or +``TERMUX_VA_DISABLE_AV1=1`` for the corresponding codec. The values ``1``, +``true``, and ``on`` enable a switch. A hidden profile is not advertised by +VA-API, so applications can use their native software decoder instead of +sending that format through the bridge. This is useful when Android's +MediaCodec exposes only a software component for a format, or when a codec +needs to be disabled for compatibility testing. + +Socket location +--------------- + +Default endpoint (container view): ``/tmp/termux-va/termux-va.sock`` - +the same directory as the Termux daemon's ``$TMPDIR/termux-va/`` through +the shared tmp mount. Both ends understand ``TERMUX_VA_SOCKET`` (full +socket path) and ``TERMUX_VA_SOCKET_DIR`` (directory), so one setting +covers the daemon and the bridge; Android system properties are accepted +as a fallback through Mesa's ``os_get_option``. + +Underlying screen +----------------- + +The decode surfaces live on a screen created by the bridge before the +frontend asks for one. ``TERMUX_VA_GPU_BACKEND`` selects how: + +``auto`` (default) tries the stock loader first and falls back to llvmpipe. It does not try the KGSL alias automatically because environments that expose a display DRM node may not have a usable Vulkan or stock DRM path. ``kgsl`` explicitly selects the fork's KGSL Freedreno alias: GPU submission goes to ``/dev/kgsl-3d0`` while the handed fd stays the control/identity fd. The same selection is propagated to Mesa's EGL and GLX loaders, so callers do not need to add ``MESA_LOADER_DRIVER_OVERRIDE=kgsl`` or ``FD_FORCE_KGSL=1``. ``sw`` forces llvmpipe for setups without GPU access; the VA decode paths used by vainfo and ffmpeg work without a GPU. ``drm`` selects the stock loader only. + +PRoot containers +---------------- + +A PRoot container can expose ``/dev/kgsl-3d0`` while exposing no usable DRM +render node. With ``TERMUX_VA_GPU_BACKEND=kgsl`` the bridge opens KGSL itself +when the display backend supplies no fd. It also uses one linear dma-buf for +both NV12 planes when no DRM render node is present, because Chromium's native +pixmap importer currently accepts only one dma-buf for this format. Override +this choice with ``DMD_VA_CONTIGUOUS_DMABUF`` or +``TERMUX_VA_CONTIGUOUS_DMABUF`` when needed. + +Chromium's native Wayland Ozone backend normally requires a DRM render node +for its GPU process, independently of VA-API. For a DRM-less PRoot desktop, +enable ``-Dtermux-va-wayland-shim=enabled`` and load the installed +``libtva_drm_shim_wayland.so`` into Chromium with ``LD_PRELOAD``. The shim +maps Chromium's DRM discovery calls to ``/dev/kgsl-3d0``; it does not replace +the Mesa KGSL backend or create a DRM device for other applications. Launch +Chromium with the native Wayland platform, ``--render-node-override=/dev/kgsl-3d0`` +and ``--hardware-video-device-path=/dev/kgsl-3d0``: + +.. code-block:: sh + + export WAYLAND_DISPLAY=wayland-0 + export XDG_RUNTIME_DIR=/run/user/$(id -u) + export LIBVA_DRIVER_NAME=termuxva + export TERMUX_VA_BRIDGE=1 + export TERMUX_VA_GPU_BACKEND=kgsl + export LD_PRELOAD=/usr/lib/aarch64-linux-gnu/libtva_drm_shim_wayland.so + google-chrome --ozone-platform=wayland --no-sandbox --use-gl=angle \ + --use-angle=gles --render-node-override=/dev/kgsl-3d0 \ + --hardware-video-device-path=/dev/kgsl-3d0 \ + --enable-features=VaapiIgnoreDriverChecks,AcceleratedVideoDecoder,AcceleratedVideoDecodeLinuxGL,AcceleratedVideoDecodeLinuxZeroCopyGL + +The standard libva DRM backend also rejects a KGSL fd; use this Mesa build's +bridge driver or an equivalent KGSL-aware compatibility layer. + +Data path +--------- + +- vaRenderPicture: the frontend parses the VA buffers and hands the + bridge slice data that already carries H.264/HEVC start codes + (parameter sets arrive as slice data buffers). +- vaEndPicture: the bridge sends the complete access unit as one daemon + input unit and associates one pending fence with the picture. The normal + pending depth defaults to 6 and may grow while the decoder reorders output; + SHM mode clamps it to 8 to stay within the daemon's slot pool. +- vaSyncSurface: the bridge waits for the frame tagged with the + picture's unit index, stages it, and copies the visible (cropped) + region into the surface's plane resources, honoring the decoder's + stride/slice-height geometry (Venus aligns buffers to 128x32). + +Frames come back inline on the socket or zero-copy through a memfd slot +pool handed over via SCM_RIGHTS (disable with ``DMD_WANT_SHM=0``). +``vaDeriveImage`` is not available (plane resources are separate +textures); CPU consumers can use ``vaGetImage``. + +Troubleshooting +--------------- + +- vainfo shows no profiles: the daemon is not running, or the consumer + environment lacks ``LIBVA_DRIVER_NAME=termuxva`` / bridge activation. +- "endpoint inode mismatch": the socket path resolves to a stale socket + (a single socket FILE was bind-mounted and the daemon restarted). + Mount the socket DIRECTORY instead - the daemon replaces the socket + file on every start, only the directory inode is stable. +- Black frames after a seek: a drain was triggered; should not happen in + steady playback - reproduce with ``DMD_VA_LOG=1`` and the daemon's + ``-v`` log. diff --git a/meson.build b/meson.build index b8da69e9a11a..8b716424db31 100644 --- a/meson.build +++ b/meson.build @@ -859,9 +859,10 @@ _va_drivers = [ ] allow_fallback_for_libva = get_option('allow-fallback-for').contains('libva') +_termux_va_bridge = get_option('termux-va-bridge') _va = get_option('gallium-va') \ - .require(_va_drivers.contains(true), - error_message : 'VA state tracker requires at least one of the following gallium drivers: r600, radeonsi, nouveau, d3d12 (with option gallium-d3d12-video), virgl.') + .require(_va_drivers.contains(true) or not _termux_va_bridge.disabled(), + error_message : 'VA state tracker requires at least one of the following gallium drivers: r600, radeonsi, nouveau, d3d12 (with option gallium-d3d12-video), virgl, or -Dtermux-va-bridge for the termux-va bridge.') _dep_va_name = host_machine.system() == 'windows' ? 'libva-win32' : 'libva' dep_va = dependency( _dep_va_name, version : '>= 1.8.0', @@ -884,6 +885,24 @@ if dep_va.found() dependencies: dep_va_headers).split('.') endif with_gallium_va = dep_va.found() +# The termux-va bridge is compiled into the VA frontend whenever it is not +# explicitly disabled; activation is decided at runtime by the TERMUX_VA_* +# environment variables (see src/gallium/frontends/va/tva_bridge.c). +with_termux_va_bridge = with_gallium_va and not _termux_va_bridge.disabled() + +_termux_va_wayland_shim = get_option('termux-va-wayland-shim') +if _termux_va_wayland_shim.enabled() + if host_machine.system() != 'linux' + error('termux-va-wayland-shim requires a Linux host') + endif + if not with_platform_wayland + error('termux-va-wayland-shim requires the Wayland platform') + endif + if not dep_libdrm.found() + error('termux-va-wayland-shim requires libdrm development headers') + endif +endif +with_termux_va_wayland_shim = _termux_va_wayland_shim.enabled() va_drivers_path = get_option('va-libs-path') if va_drivers_path == '' diff --git a/meson.options b/meson.options index 93a60e3047b9..3471a6505e77 100644 --- a/meson.options +++ b/meson.options @@ -108,6 +108,20 @@ option( description : 'enable gallium va frontend.', ) +option( + 'termux-va-bridge', + type : 'feature', + value : 'auto', + description : 'build the termux-va bridge into the VA frontend (VA decode forwarded over a Unix socket to the Termux termux-va daemon; runtime-gated by the TERMUX_VA_* environment variables)', +) + +option( + 'termux-va-wayland-shim', + type : 'feature', + value : 'disabled', + description : 'build the LD_PRELOAD DRM compatibility shim for Chromium Wayland in DRM-less KGSL PRoot containers', +) + option( 'gallium-mediafoundation', type : 'feature', diff --git a/src/egl/drivers/dri2/platform_x11.c b/src/egl/drivers/dri2/platform_x11.c index a31a9d353abc..78711177f936 100644 --- a/src/egl/drivers/dri2/platform_x11.c +++ b/src/egl/drivers/dri2/platform_x11.c @@ -591,6 +591,8 @@ dri2_x11_add_configs_for_visuals(struct dri2_egl_display *dri2_dpy, xcb_depth_iterator_t d; xcb_visualtype_t *visuals; EGLint surface_type; + const bool kgsl = dri2_dpy->driver_name && + strcmp(dri2_dpy->driver_name, "kgsl") == 0; d = xcb_screen_allowed_depths_iterator(dri2_dpy->screen); @@ -607,7 +609,11 @@ dri2_x11_add_configs_for_visuals(struct dri2_egl_display *dri2_dpy, visuals = xcb_depth_visuals(d.data); for (int i = 0; i < xcb_depth_visuals_length(d.data); i++) { - if (class_added[visuals[i]._class]) + /* XWayland may expose several equivalent visuals on KGSL. ANGLE + * clients can select one through GLX and then require an EGLConfig + * with that exact visual ID when creating an EGL window surface. + */ + if (!kgsl && class_added[visuals[i]._class]) continue; class_added[visuals[i]._class] = EGL_TRUE; diff --git a/src/egl/main/eglapi.c b/src/egl/main/eglapi.c index 4ff1a9d0a0ab..0d7d643619ea 100644 --- a/src/egl/main/eglapi.c +++ b/src/egl/main/eglapi.c @@ -684,8 +684,13 @@ eglInitialize(EGLDisplay dpy, EGLint *major, EGLint *minor) "Found 'LIBGL_ALWAYS_SOFTWARE' set, will use a CPU renderer"); const char *env = os_get_option("MESA_LOADER_DRIVER_OVERRIDE"); - disp->Options.Zink = !env || !strcmp(env, "zink"); - disp->Options.Kgsl = env && !strcmp(env, "kgsl"); + const char *backend = os_get_option("TERMUX_VA_GPU_BACKEND"); + const bool kgsl_backend = backend && !strcmp(backend, "kgsl"); + if (!env && kgsl_backend && + setenv("MESA_LOADER_DRIVER_OVERRIDE", "kgsl", 0) == 0) + env = os_get_option("MESA_LOADER_DRIVER_OVERRIDE"); + disp->Options.Zink = !kgsl_backend && (!env || !strcmp(env, "zink")); + disp->Options.Kgsl = kgsl_backend || (env && !strcmp(env, "kgsl")); const char *gallium_hud_env = os_get_option("GALLIUM_HUD"); disp->Options.GalliumHudWarn = diff --git a/src/freedreno/drm/freedreno_bo.c b/src/freedreno/drm/freedreno_bo.c index 6eda8101e46c..779d95529200 100644 --- a/src/freedreno/drm/freedreno_bo.c +++ b/src/freedreno/drm/freedreno_bo.c @@ -268,6 +268,24 @@ fd_bo_from_dmabuf(struct fd_device *dev, int fd) return dev->funcs->bo_from_dmabuf(dev, fd); } +int +fd_bo_sync_to_gpu(struct fd_bo *bo) +{ + if (!bo) + return 0; + + /* Uploaders commonly suballocate small command, vertex, and constant + * buffers from a KGSL heap block. The suballocation has no kernel handle + * of its own, so synchronize the real backing BO instead. */ + if (suballoc_bo(bo)) + bo = fd_bo_heap_block(bo); + + if (!bo->funcs->sync_to_gpu) + return 0; + + return bo->funcs->sync_to_gpu(bo); +} + struct fd_bo * fd_bo_from_name(struct fd_device *dev, uint32_t name) { @@ -844,4 +862,3 @@ fd_bo_state(struct fd_bo *bo) return FD_BO_STATE_BUSY; } - diff --git a/src/freedreno/drm/freedreno_drmif.h b/src/freedreno/drm/freedreno_drmif.h index 930a476ab039..4fd1436436d9 100644 --- a/src/freedreno/drm/freedreno_drmif.h +++ b/src/freedreno/drm/freedreno_drmif.h @@ -316,6 +316,7 @@ int fd_bo_get_name(struct fd_bo *bo, uint32_t *name); uint32_t fd_bo_handle(struct fd_bo *bo); int fd_bo_dmabuf_drm(struct fd_bo *bo); int fd_bo_dmabuf(struct fd_bo *bo); +int fd_bo_sync_to_gpu(struct fd_bo *bo); uint32_t fd_bo_size(struct fd_bo *bo); void *fd_bo_map(struct fd_bo *bo); void fd_bo_upload(struct fd_bo *bo, void *src, unsigned off, unsigned len); diff --git a/src/freedreno/drm/freedreno_priv.h b/src/freedreno/drm/freedreno_priv.h index 19c9d37f40e6..69b1608a2cc9 100644 --- a/src/freedreno/drm/freedreno_priv.h +++ b/src/freedreno/drm/freedreno_priv.h @@ -424,6 +424,7 @@ struct fd_bo_funcs { uint64_t (*iova)(struct fd_bo *bo); void (*set_name)(struct fd_bo *bo, const char *fmt, va_list ap); int (*dmabuf)(struct fd_bo *bo); + int (*sync_to_gpu)(struct fd_bo *bo); /** * Optional hook that is called before ->destroy(). In the case of diff --git a/src/freedreno/drm/kgsl/kgsl_bo.c b/src/freedreno/drm/kgsl/kgsl_bo.c index c676b8197a5f..2302f84b9caf 100644 --- a/src/freedreno/drm/kgsl/kgsl_bo.c +++ b/src/freedreno/drm/kgsl/kgsl_bo.c @@ -115,6 +115,48 @@ static int kgsl_bo_dmabuf(struct fd_bo *bo) { return os_dupfd_cloexec(kgsl_bo->import_fd); } +static int +kgsl_bo_sync_to_gpu(struct fd_bo *bo) +{ + /* The legacy GPUOBJ_SYNC path is not implemented consistently by all + * Android KGSL kernels. Try the explicit cache ioctl first; this is the + * operation that tells the dma-buf exporter to clean CPU lines for a GPU + * consumer. */ + struct kgsl_gpumem_sync_cache cache = { + .gpuaddr = 0, + .id = bo->handle, + .op = KGSL_GPUMEM_CACHE_FLUSH, + .offset = 0, + .length = bo->size, + }; + int cache_ret = kgsl_pipe_safe_ioctl(bo->dev->fd, + IOCTL_KGSL_GPUMEM_SYNC_CACHE, + &cache); + + struct kgsl_gpuobj_sync_obj sync_obj = { + .offset = 0, + .length = bo->size, + .id = bo->handle, + .op = KGSL_GPUMEM_CACHE_FLUSH, + }; + struct kgsl_gpuobj_sync sync = { + .objs = (uintptr_t)&sync_obj, + .obj_len = sizeof(sync_obj), + .count = 1, + }; + + int obj_ret = kgsl_pipe_safe_ioctl(bo->dev->fd, + IOCTL_KGSL_GPUOBJ_SYNC, &sync); + if (getenv("DMD_VA_LOG")) + fprintf(stderr, "kgsl: sync-to-gpu id=%u size=%u cache=%d/%d obj=%d/%d\n", + bo->handle, bo->size, cache_ret, cache_ret ? errno : 0, + obj_ret, obj_ret ? errno : 0); + + if (cache_ret == 0 || obj_ret == 0) + return 0; + return obj_ret; +} + static const struct fd_bo_funcs bo_funcs = { .iova = kgsl_bo_iova, .set_name = kgsl_bo_set_name, @@ -124,6 +166,7 @@ static const struct fd_bo_funcs bo_funcs = { .cpu_prep = kgsl_bo_cpu_prep, .destroy = kgsl_bo_destroy, .dmabuf = kgsl_bo_dmabuf, + .sync_to_gpu = kgsl_bo_sync_to_gpu, }; /* Size is not used by KGSL */ @@ -190,6 +233,28 @@ kgsl_bo_from_dmabuf(struct fd_device *dev, int fd) kgsl_bo->bo_type = KGSL_BO_IMPORT; kgsl_bo->import_fd = os_dupfd_cloexec(fd); + /* A dma-buf may have been populated through a cached CPU mapping in + * another API (for example the termux-va VA bridge). KGSL does not + * infer that CPU-to-GPU transition from the dma-buf import itself. Flush + * the imported object before the first GPU read, matching Turnip's KGSL + * import path. Keep import successful on kernels which do not implement + * GPUOBJ_SYNC; those kernels still retain the historical behaviour. */ + struct kgsl_gpuobj_sync_obj sync_obj = { + .offset = 0, + .length = bo->size, + .id = bo->handle, + .op = KGSL_GPUMEM_CACHE_FLUSH, + }; + struct kgsl_gpuobj_sync sync = { + .objs = (uintptr_t)&sync_obj, + .obj_len = sizeof(sync_obj), + .count = 1, + }; + if (kgsl_pipe_safe_ioctl(dev->fd, IOCTL_KGSL_GPUOBJ_SYNC, &sync) != 0 && + getenv("DMD_VA_LOG")) + fprintf(stderr, "kgsl: dma-buf GPU cache sync failed id=%u errno=%d\n", + bo->handle, errno); + return bo; } diff --git a/src/freedreno/vulkan/tu_clear_blit.cc b/src/freedreno/vulkan/tu_clear_blit.cc index af49cf0c34c0..0acb82ed2e02 100644 --- a/src/freedreno/vulkan/tu_clear_blit.cc +++ b/src/freedreno/vulkan/tu_clear_blit.cc @@ -26,6 +26,9 @@ #include "tu_lrz.h" #include "tu_tracepoints.h" +#include +#include + static const VkOffset2D blt_no_coord = { ~0, ~0 }; /* The helpers below quantize floats to match shader export behavior and avoid @@ -2693,6 +2696,29 @@ tu_copy_buffer_to_image(struct tu_cmd_buffer *cmd, uint32_t pitch = src_width * block_size; uint32_t layer_size = src_height * pitch; + if (getenv("TU_KGSL_DEBUG_COPY")) { + fprintf(stderr, + "tu upload fmt=%d aspect=%#x srcfmt=%d dstfmt=%d img=%llu " + "extent=%ux%u off=%d,%d src_iova=%#llx img_iova=%#llx " + "img_size=%llu row=%u height=%u boff=%llu\n", + dst_image->vk.format, info->imageSubresource.aspectMask, + src_format, dst_format, + (unsigned long long)dst_image->total_size, + extent.width, extent.height, offset.x, offset.y, + (unsigned long long)vk_buffer_address(&src_buffer->vk, + info->bufferOffset), + (unsigned long long)dst_image->iova, + (unsigned long long)dst_image->total_size, + info->bufferRowLength, info->bufferImageHeight, + (unsigned long long)info->bufferOffset); + if (src_buffer->bo->map) { + const uint8_t *p = (const uint8_t *)src_buffer->bo->map + + info->bufferOffset; + fprintf(stderr, "tu upload src bytes=%02x %02x %02x %02x %02x %02x %02x %02x\n", + p[0], p[1], p[2], p[3], p[4], p[5], p[6], p[7]); + } + } + ops->setup(cmd, cs, src_format, dst_format, info->imageSubresource.aspectMask, blit_param, false, dst_image->layout[0].ubwc, (VkSampleCountFlagBits) dst_image->layout[0].nr_samples, @@ -2899,7 +2925,10 @@ tu_copy_image_to_buffer(struct tu_cmd_buffer *cmd, /* note: could use "R8_UNORM" when no UBWC */ unsigned blit_param = 0; - if (dst_format == PIPE_FORMAT_Y8_UNORM) { + if (dst_format == PIPE_FORMAT_Y8_UNORM || + (getenv("TU_KGSL_FORCE_R3D_COPY") && + (dst_format == PIPE_FORMAT_R8_UNORM || + dst_format == PIPE_FORMAT_R8G8_UNORM))) { ops = &r3d_ops; blit_param = R3D_COPY; } @@ -2915,6 +2944,29 @@ tu_copy_image_to_buffer(struct tu_cmd_buffer *cmd, uint32_t pitch = dst_width * block_size; uint32_t layer_size = pitch * dst_height; + if (getenv("TU_KGSL_DEBUG_COPY")) { + unsigned plane = tu6_plane_index(src_image->vk.format, + info->imageSubresource.aspectMask); + const struct fdl_layout *layout = &src_image->layout[plane]; + fprintf(stderr, + "tu copy fmt=%d aspect=%#x plane=%u srcfmt=%d dstfmt=%d " + "img=%llu extent=%ux%u off=%d,%d layout=%ux%u pitch=%u " + "layer=%llu iova=%#llx dst_iova=%#llx dst_size=%llu " + "row=%u height=%u boff=%llu ops=%s\n", + src_image->vk.format, info->imageSubresource.aspectMask, plane, + src_format, dst_format, (unsigned long long)src_image->total_size, + extent.width, extent.height, offset.x, offset.y, + layout->width0, layout->height0, fdl_pitch(layout, 0), + (unsigned long long)fdl_layer_stride(layout, 0), + (unsigned long long)src_image->iova, + (unsigned long long)vk_buffer_address(&dst_buffer->vk, + info->bufferOffset), + (unsigned long long)dst_buffer->bo->size, + info->bufferRowLength, info->bufferImageHeight, + (unsigned long long)info->bufferOffset, + ops == &r3d_ops ? "r3d" : "r2d"); + } + handle_buffer_unaligned_store(cmd, vk_buffer_address(&dst_buffer->vk, info->bufferOffset), layer_size * layers, unaligned_store); @@ -2926,6 +2978,15 @@ tu_copy_image_to_buffer(struct tu_cmd_buffer *cmd, tu_image_view_copy(&src, src_image, src_format, &info->imageSubresource, offset.z); + if (getenv("TU_KGSL_DEBUG_COPY")) { + fprintf(stderr, + "tu copy view base=%#llx off=%u pitch=%u layer=%u size=%ux%u " + "format=%d ubwc=%d\n", + (unsigned long long)src.base_addr, src.offset, src.pitch, + src.layer_size, src.width, src.height, src.format, + src.ubwc_enabled); + } + for (uint32_t i = 0; i < layers; i++) { ops->src(cmd, cs, &src, i, VK_FILTER_NEAREST, dst_format); diff --git a/src/freedreno/vulkan/tu_cmd_buffer.cc b/src/freedreno/vulkan/tu_cmd_buffer.cc index 489f3e090db3..1ba151195384 100644 --- a/src/freedreno/vulkan/tu_cmd_buffer.cc +++ b/src/freedreno/vulkan/tu_cmd_buffer.cc @@ -30,6 +30,10 @@ #include "tu_tracepoints.h" #include "tu_trace_bin_layout.h" +#include +#include +#include + enum tu_cmd_buffer_status { TU_CMD_BUFFER_STATUS_IDLE = 0, TU_CMD_BUFFER_STATUS_ACTIVE = 1, @@ -5821,6 +5825,14 @@ tu_flush_for_access(struct tu_cache_state *cache, cache->flush_bits |= flush_bits; cache->pending_flush_bits &= ~flush_bits; + + if (getenv("TU_KGSL_DEBUG_BARRIER")) { + fprintf(stderr, + "tu access src=%#x dst=%#x add=%#x flush=%#x pending=%#x\n", + (unsigned)src_mask, (unsigned)dst_mask, (unsigned)flush_bits, + (unsigned)cache->flush_bits, + (unsigned)cache->pending_flush_bits); + } } /* When translating Vulkan access flags to which cache is accessed @@ -10184,6 +10196,18 @@ tu_barrier(struct tu_cmd_buffer *cmd, struct tu_cache_state *cache = cmd->state.pass ? &cmd->state.renderpass_cache : &cmd->state.cache; + if (getenv("TU_KGSL_DEBUG_BARRIER")) { + fprintf(stderr, + "tu barrier srcStage=%#" PRIx64 " dstStage=%#" PRIx64 + " src=%#x dst=%#x no_sync=%d pass=%d gmem=%d preflush=%#x" + " prepen=%#x\n", + (uint64_t)srcStage, (uint64_t)dstStage, + (unsigned)src_flags, (unsigned)dst_flags, no_sync, + cmd->state.pass != NULL, gmem, + (unsigned)cache->flush_bits, + (unsigned)cache->pending_flush_bits); + } + /* a750 has a HW bug where writing a UBWC compressed image with a compute * shader followed by reading it as a texture (or readonly image) requires * a CACHE_CLEAN event. Some notes about this bug: diff --git a/src/freedreno/vulkan/tu_device.cc b/src/freedreno/vulkan/tu_device.cc index 83baf5753165..20d2d9dbf1b0 100644 --- a/src/freedreno/vulkan/tu_device.cc +++ b/src/freedreno/vulkan/tu_device.cc @@ -9,6 +9,8 @@ #include "tu_device.h" +#include + #include "drm-uapi/drm_fourcc.h" #include "git_sha1.h" #include "perfcntrs/freedreno_perfcntr.h" @@ -252,7 +254,11 @@ get_device_extensions(const struct tu_physical_device *device, .KHR_external_memory = true, .KHR_external_memory_fd = true, .KHR_external_semaphore = true, - .KHR_external_semaphore_fd = true, + /* KGSL exposes binary SYNC_FD fences only. OPAQUE_FD timeline + * semaphores are unsupported, yet FFmpeg treats this extension as + * evidence that they are available and cannot import dma-heap frames. + */ + .KHR_external_semaphore_fd = !is_kgsl(device->instance), .KHR_format_feature_flags2 = true, .KHR_fragment_shading_rate = device->info->props.has_attachment_shading_rate, .KHR_get_memory_requirements2 = true, @@ -1792,6 +1798,17 @@ tu_physical_device_init(struct tu_physical_device *device, device->level1_dcache_size = util_cache_granularity(); device->has_cached_non_coherent_memory = device->level1_dcache_size > 0 && !DETECT_ARCH_ARM; + + /* KGSL on some Android kernels reports IOCOHERENT support even though + * userspace mappings still require explicit cache maintenance. Keep the + * normal capability probing as the default, but provide a device-side + * switch while diagnosing such stacks. This makes Vulkan expose a cached + * non-coherent memory type and routes map readback through the invalidate + * path instead of relying on the (incorrect) HOST_COHERENT declaration. */ + if (getenv("TU_KGSL_FORCE_NONCOHERENT")) { + device->has_cached_coherent_memory = false; + device->has_cached_non_coherent_memory = device->level1_dcache_size > 0; + } device->preferred_uncached_as_cached_index = -1; device->memory.type_count = 1; diff --git a/src/freedreno/vulkan/tu_knl_kgsl.cc b/src/freedreno/vulkan/tu_knl_kgsl.cc index 2fa2ba34ef72..4986ca2bb563 100644 --- a/src/freedreno/vulkan/tu_knl_kgsl.cc +++ b/src/freedreno/vulkan/tu_knl_kgsl.cc @@ -8,6 +8,9 @@ #include #include #include +#include +#include +#include #include #include @@ -47,6 +50,32 @@ safe_ioctl(int fd, unsigned long request, void *arg) return ret; } +/* Imported dma-bufs are backed by cached system memory on KGSL. A producer + * in another API (for example the termux-va Gallium bridge) can have dirty + * CPU cache lines when Turnip imports the object. Synchronize the object + * before exposing it to GPU commands; Vulkan's mapped-memory entrypoints are + * not involved for an externally imported image. + */ +static int +kgsl_sync_imported_bo_to_gpu(struct tu_device *dev, uint32_t id, + uint64_t size) +{ + struct kgsl_gpuobj_sync_obj obj = { + .offset = 0, + .length = size, + .id = id, + .op = KGSL_GPUMEM_CACHE_FLUSH, + }; + struct kgsl_gpuobj_sync sync = { + .objs = (uintptr_t)&obj, + .obj_len = sizeof(obj), + .count = 1, + }; + + return safe_ioctl(dev->physical_device->local_fd, + IOCTL_KGSL_GPUOBJ_SYNC, &sync); +} + static int kgsl_submitqueue_new(struct tu_device *dev, struct tu_queue *queue) { @@ -355,6 +384,11 @@ kgsl_bo_init(struct tu_device *dev, .base = base, }; + if (getenv("TU_KGSL_DEBUG_IMPORT")) + fprintf(stderr, "tu kgsl new id=%u size=%llu iova=%#llx flags=%#x\n", + bo->gem_handle, (unsigned long long)bo->size, + (unsigned long long)bo->iova, (unsigned)mem_property); + tu_dump_bo_init(dev, bo); VkResult result = VK_SUCCESS; @@ -425,6 +459,11 @@ kgsl_bo_init_dmabuf(struct tu_device *dev, .shared_fd = os_dupfd_cloexec(fd), }; + if (getenv("TU_KGSL_DEBUG_IMPORT")) + fprintf(stderr, "tu kgsl dmabuf id=%u size=%llu iova=%#llx fd=%d\n", + bo->gem_handle, (unsigned long long)bo->size, + (unsigned long long)bo->iova, fd); + struct stat st; if (fstat(fd, &st) == 0) /* Use the inode number as the unique ID, but set the MSB to avoid @@ -434,6 +473,30 @@ kgsl_bo_init_dmabuf(struct tu_device *dev, tu_dump_bo_init(dev, bo); + /* The dma-buf may have been written through a cached CPU mapping by the + * producer. Clean it before the first Vulkan GPU read. */ + int sync_ret = kgsl_sync_imported_bo_to_gpu(dev, bo->gem_handle, bo->size); + if (sync_ret != 0) + mesa_logw("KGSL cache sync for imported dma-buf failed: %s\n", + strerror(errno)); + + if (getenv("TU_KGSL_DEBUG_IMPORT")) { + void *map = mmap(NULL, MIN2((uint64_t)4096, bo->size), PROT_READ, + MAP_SHARED, fd, 0); + if (map == MAP_FAILED) { + fprintf(stderr, "tu kgsl import id=%u size=%llu" + " sync=%d errno=%d mmap failed\n", + bo->gem_handle, (unsigned long long)bo->size, sync_ret, errno); + } else { + const uint8_t *p = (const uint8_t *)map; + fprintf(stderr, "tu kgsl import id=%u size=%llu" + " sync=%d bytes=%02x %02x %02x %02x %02x %02x %02x %02x\n", + bo->gem_handle, (unsigned long long)bo->size, sync_ret, + p[0], p[1], p[2], p[3], p[4], p[5], p[6], p[7]); + munmap(map, MIN2((uint64_t)4096, bo->size)); + } + } + *out_bo = bo; return VK_SUCCESS; @@ -1352,6 +1415,20 @@ kgsl_submit_add_entries(struct tu_device *device, void *_submit, .flags = KGSL_CMDLIST_IB, .id = entries[i].bo->gem_handle, }; + if (getenv("TU_KGSL_DEBUG_SUBMIT")) + fprintf(stderr, "kgsl cmd gpu=%#llx size=%llu id=%u off=%llu\n", + (unsigned long long)cmds[i].gpuaddr, + (unsigned long long)cmds[i].size, + cmds[i].id, (unsigned long long)cmds[i].offset); + if (getenv("TU_KGSL_DEBUG_SUBMIT") && entries[i].bo->map) { + const uint32_t *p = (const uint32_t *)entries[i].bo->map + + entries[i].offset / 4; + unsigned n = MIN2(entries[i].size / 4, 1000); + fprintf(stderr, "kgsl ib:"); + for (unsigned j = 0; j < n; j++) + fprintf(stderr, " %08x", p[j]); + fprintf(stderr, "\n"); + } } } @@ -1432,6 +1509,19 @@ kgsl_queue_submit(struct tu_queue *queue, void *_submit, struct tu_kgsl_queue_submit *submit = (struct tu_kgsl_queue_submit *)_submit; + if (getenv("TU_KGSL_DEBUG_SUBMIT")) { + fprintf(stderr, "kgsl submit begin queue=%u prior_ts=%d waits=%u signals=%u cmds=%u binds=%u\n", + queue->msm_queue_id, p_atomic_read(&queue->fence), wait_count, + signal_count, submit->commands.size, submit->bind_cmds.size); + for (uint32_t i = 0; i < wait_count; i++) { + const struct kgsl_syncobj *wait = + &container_of(waits[i].sync, struct vk_kgsl_syncobj, vk)->syncobj; + fprintf(stderr, "kgsl submit wait[%u] state=%d queue=%u ts=%u fd=%d\n", + i, wait->state, wait->queue ? wait->queue->msm_queue_id : 0, + wait->timestamp, wait->fd); + } + } + #if HAVE_PERFETTO uint64_t start_ts = tu_perfetto_begin_submit(); #endif @@ -1548,6 +1638,12 @@ kgsl_queue_submit(struct tu_queue *queue, void *_submit, assert(wait_sync.state != KGSL_SYNCOBJ_STATE_UNSIGNALED); // Would wait forever + if (getenv("TU_KGSL_DEBUG_SUBMIT")) + fprintf(stderr, "kgsl submit merged-wait state=%d queue=%u ts=%u fd=%d\n", + wait_sync.state, + wait_sync.queue ? wait_sync.queue->msm_queue_id : 0, + wait_sync.timestamp, wait_sync.fd); + struct kgsl_cmd_syncpoint_timestamp ts; struct kgsl_cmd_syncpoint_fence fn; struct kgsl_command_syncpoint sync = { 0 }; @@ -1607,6 +1703,9 @@ kgsl_queue_submit(struct tu_queue *queue, void *_submit, IOCTL_KGSL_GPU_COMMAND, &req); timestamp = req.timestamp; + if (getenv("TU_KGSL_DEBUG_SUBMIT")) + fprintf(stderr, "kgsl submit ret=%d errno=%d ts=%u cmds=%u\n", + ret, ret ? errno : 0, timestamp, req.numcmds); } else { /* kgsl doesn't support multiple bind commands at once */ uint32_t i = 0; diff --git a/src/gallium/auxiliary/vl/vl_compositor_gfx.c b/src/gallium/auxiliary/vl/vl_compositor_gfx.c index c36937d52330..abcf52f05803 100644 --- a/src/gallium/auxiliary/vl/vl_compositor_gfx.c +++ b/src/gallium/auxiliary/vl/vl_compositor_gfx.c @@ -26,6 +26,7 @@ **************************************************************************/ #include +#include #include "util/compiler.h" #include "pipe/p_context.h" @@ -417,8 +418,16 @@ create_frag_shader_rgba(struct vl_compositor *c) /* * fragment = tex(tc, sampler) */ - ureg_TEX(shader, texel, TGSI_TEXTURE_2D, tc, sampler); - ureg_MUL(shader, fragment, ureg_src(texel), color); + if (getenv("DMD_VA_PROBE")) { + /* Temporary diagnostic: exercise the complete draw path without a + * texture fetch. A non-zero target proves that the failure is in + * source visibility or texture sampling rather than framebuffer setup. + */ + ureg_MOV(shader, fragment, ureg_imm4f(shader, 1.0f, 0.0f, 0.0f, 1.0f)); + } else { + ureg_TEX(shader, texel, TGSI_TEXTURE_2D, tc, sampler); + ureg_MUL(shader, fragment, ureg_src(texel), color); + } ureg_END(shader); return ureg_create_shader_and_destroy(shader, c->pipe); @@ -596,6 +605,7 @@ static void gen_vertex_data(struct vl_compositor *c, struct vl_compositor_state *s, struct u_rect *dirty, struct pipe_resource **releasebuf) { struct vertex2f *vb; + struct vertex2f *base; unsigned i; assert(c); @@ -607,6 +617,7 @@ gen_vertex_data(struct vl_compositor *c, struct vl_compositor_state *s, struct u &c->vertex_buf.buffer_offset, &c->vertex_buf.buffer.resource, releasebuf, (void **)&vb); + base = vb; for (i = 0; i < VL_COMPOSITOR_MAX_LAYERS; i++) { if (s->used_layers & (1 << i)) { @@ -637,6 +648,15 @@ gen_vertex_data(struct vl_compositor *c, struct vl_compositor_state *s, struct u } } + if (getenv("DMD_VA_PROBE") && (s->used_layers & 1)) + fprintf(stderr, "tva-vl layer dst=%g,%g-%g,%g vp=%g,%g+%g,%g vertices p0=%g,%g p1=%g,%g p2=%g,%g p3=%g,%g\n", + s->layers[0].dst.tl.x, s->layers[0].dst.tl.y, + s->layers[0].dst.br.x, s->layers[0].dst.br.y, + s->layers[0].viewport.scale[0], s->layers[0].viewport.scale[1], + s->layers[0].viewport.translate[0], s->layers[0].viewport.translate[1], + base[0].x, base[0].y, base[5].x, base[5].y, + base[10].x, base[10].y, base[15].x, base[15].y); + u_upload_unmap(c->pipe->stream_uploader); } @@ -683,7 +703,11 @@ draw_layers(struct vl_compositor *c, struct vl_compositor_state *s, struct u_rec c->pipe->set_sampler_views(c->pipe, MESA_SHADER_FRAGMENT, 0, num_sampler_views, 0, samplers); - util_draw_arrays(c->pipe, MESA_PRIM_QUADS, vb_index * 4, 4); + /* Freedreno's hardware primitive table does not implement + * MESA_PRIM_QUADS (it maps to DI_PT_NONE). A triangle fan keeps + * both triangles wound consistently on Adreno while retaining the + * four-vertex compositor layout. */ + util_draw_arrays(c->pipe, MESA_PRIM_TRIANGLE_FAN, vb_index * 4, 4); vb_index++; if (dirty) { @@ -732,6 +756,13 @@ vl_compositor_gfx_render(struct vl_compositor_state *s, dirty_area->x1 = dirty_area->y1 = VL_COMPOSITOR_MIN_DIRTY; } + if (getenv("DMD_VA_CLEAR_BEFORE_DRAW")) { + union pipe_color_union color = { .f = { 0.0f, 1.0f, 0.0f, 1.0f } }; + c->pipe->clear_render_target(c->pipe, dst_surface, &color, + 0, 0, c->fb_state.width, + c->fb_state.height, false); + } + c->pipe->set_framebuffer_state(c->pipe, &c->fb_state); c->pipe->bind_vs_state(c->pipe, c->vs); c->pipe->bind_vertex_elements_state(c->pipe, c->vertex_elems_state); diff --git a/src/gallium/auxiliary/vl/vl_compositor_proc.c b/src/gallium/auxiliary/vl/vl_compositor_proc.c index 7de461e1d157..c02969185bcc 100644 --- a/src/gallium/auxiliary/vl/vl_compositor_proc.c +++ b/src/gallium/auxiliary/vl/vl_compositor_proc.c @@ -4,6 +4,14 @@ * SPDX-License-Identifier: MIT */ +#include +#include +#include +#include +#include + +#include "util/u_sampler.h" + #include "vl_compositor_proc.h" #include "vl_compositor.h" #include "vl_video_buffer.h" @@ -16,6 +24,13 @@ struct vl_compositor_proc { struct pipe_video_buffer *target; }; +static bool +tva_native_sample_test_enabled(void) +{ + const char *e = getenv("DMD_VA_NATIVE_SAMPLE_TEST"); + return e && (!strcmp(e, "1") || !strcmp(e, "true") || !strcmp(e, "on")); +} + static void compositor_proc_destroy(struct pipe_video_codec *codec) { @@ -46,11 +61,81 @@ compositor_proc_process_frame(struct pipe_video_codec *codec, enum vl_compositor_rotation rotation; enum vl_compositor_mirror mirror; struct pipe_video_buffer *dst = proc->target; + struct pipe_context *pipe = proc->b.context; struct pipe_vpp_desc *param = (struct pipe_vpp_desc *)process_properties; enum vl_compositor_deinterlace deinterlace = VL_COMPOSITOR_NONE; bool src_yuv = util_format_is_yuv(src->buffer_format); bool dst_yuv = util_format_is_yuv(dst->buffer_format); + if (getenv("DMD_VA_PROBE")) + fprintf(stderr, "tva-proc: enter pid=%d probe=%s dst_yuv=%d dst_fmt=%s " + "cs=%d gfx=%d fs_rgba=%p cs_rgba=%p\n", + (int)getpid(), getenv("DMD_VA_PROBE"), dst_yuv, + util_format_short_name(dst->buffer_format), + proc->compositor.pipe_cs_composit_supported, + proc->compositor.pipe_gfx_supported, + proc->compositor.fs_rgba, proc->compositor.cs_rgba); + + if (getenv("DMD_VA_PROBE")) { + struct pipe_resource *src_resources[VL_NUM_COMPONENTS] = {0}; + struct pipe_surface *dst_surfaces = dst->get_surfaces(dst); + src->get_resources(src, src_resources); + fprintf(stderr, "tva-proc: process src=%s %ux%u dst=%s %ux%u src0=%p bind=%#x dst0=%p bind=%#x\n", + util_format_short_name(src->buffer_format), src->width, src->height, + util_format_short_name(dst->buffer_format), dst->width, dst->height, + (void *)src_resources[0], src_resources[0] ? src_resources[0]->bind : 0, + dst_surfaces ? (void *)dst_surfaces[0].texture : NULL, + dst_surfaces && dst_surfaces[0].texture ? dst_surfaces[0].texture->bind : 0); + fprintf(stderr, "tva-proc: regions src=%d,%d-%d,%d dst=%d,%d-%d,%d orient=%#x blend=%d/%#x alpha=%f colors=%d/%d/%d/%d/%d/%d\n", + param->src_region.x0, param->src_region.y0, + param->src_region.x1, param->src_region.y1, + param->dst_region.x0, param->dst_region.y0, + param->dst_region.x1, param->dst_region.y1, + param->orientation, param->blend.enabled, param->blend.mode, + param->blend.global_alpha, param->in_color_range, + param->out_color_range, param->in_matrix_coefficients, + param->out_matrix_coefficients, param->in_color_primaries, + param->out_color_primaries); + if (src_resources[0] && pipe->texture_map && pipe->texture_unmap) { + struct pipe_box box = { .x = 0, .y = 0, .z = 0, + .width = 8, .height = 1, .depth = 1 }; + struct pipe_transfer *transfer = NULL; + uint8_t *map = pipe->texture_map(pipe, src_resources[0], 0, + PIPE_MAP_READ, &box, &transfer); + if (map && transfer) { + fprintf(stderr, "tva-proc: src probe fmt=%s stride=%u bytes=%02x %02x %02x %02x %02x %02x %02x %02x\n", + util_format_short_name(src_resources[0]->format), transfer->stride, + map[0], map[1], map[2], map[3], map[4], map[5], map[6], map[7]); + pipe->texture_unmap(pipe, transfer); + } else { + fprintf(stderr, "tva-proc: src probe map failed res=%p\n", + (void *)src_resources[0]); + if (transfer) + pipe->texture_unmap(pipe, transfer); + } + } + if (src_resources[0] && pipe->texture_map && pipe->texture_unmap) { + struct pipe_box box = { .x = 960, .y = 540, .z = 0, + .width = 1, .height = 1, .depth = 1 }; + struct pipe_transfer *transfer = NULL; + uint8_t *map = pipe->texture_map(pipe, src_resources[0], 0, + PIPE_MAP_READ, &box, &transfer); + if (map && transfer) { + fprintf(stderr, "tva-proc: src center fmt=%s stride=%u bytes=%02x %02x %02x %02x\n", + util_format_short_name(src_resources[0]->format), transfer->stride, + map[0], map[1], map[2], map[3]); + pipe->texture_unmap(pipe, transfer); + } + } + if (pipe->screen->resource_changed) { + for (unsigned i = 0; i < VL_NUM_COMPONENTS; i++) { + if (src_resources[i]) + pipe->screen->resource_changed(pipe->screen, src_resources[i]); + } + fprintf(stderr, "tva-proc: notified resource changes\n"); + } + } + /* Subsampled formats not supported */ if (util_format_is_subsampled_422(dst->buffer_format)) return 1; @@ -59,6 +144,211 @@ compositor_proc_process_frame(struct pipe_video_codec *codec, if (!surfaces[0].texture) return 1; + /* Debug-only path used to distinguish imported-texture visibility from + * YUV shader issues on KGSL-only systems. */ + if (getenv("DMD_VA_PROBE") && !dst_yuv && + tva_native_sample_test_enabled()) { + struct pipe_resource *src_resources[VL_NUM_COMPONENTS] = {0}; + src->get_resources(src, src_resources); + if (!src_resources[0]) + return 1; + + struct pipe_resource templ = *src_resources[0]; + const bool native_rgba = true; + if (native_rgba) + templ.format = PIPE_FORMAT_B8G8R8A8_UNORM; + templ.bind = PIPE_BIND_SAMPLER_VIEW | PIPE_BIND_LINEAR; + templ.usage = PIPE_USAGE_DEFAULT; + templ.flags = 0; + templ.next = NULL; + struct pipe_resource *native = pipe->screen->resource_create( + pipe->screen, &templ); + fprintf(stderr, "tva-proc: NATIVE_SAMPLE_TEST imported=%p native=%p format=%s %ux%u\n", + (void *)src_resources[0], (void *)native, + native ? util_format_short_name(native->format) : "none", + native ? native->width0 : 0, native ? native->height0 : 0); + if (!native) + return 1; + + const unsigned native_blocksize = util_format_get_blocksize(native->format); + const unsigned native_stride = native->width0 * native_blocksize; + struct pipe_box box = { .x = 0, .y = 0, .z = 0, + .width = (int)native->width0, + .height = (int)native->height0, .depth = 1 }; + if (pipe->texture_subdata) { + uint8_t *upload = malloc((size_t)native_stride * native->height0); + if (!upload) { + pipe_resource_reference(&native, NULL); + return 1; + } + for (unsigned y = 0; y < native->height0; y++) { + uint8_t *row = upload + (size_t)y * native_stride; + for (unsigned x = 0; x < native->width0; x++) { + row[x * native_blocksize + 0] = 0x00; + row[x * native_blocksize + 1] = 0x00; + row[x * native_blocksize + 2] = 0xff; + row[x * native_blocksize + 3] = 0xff; + } + } + pipe->texture_subdata(pipe, native, 0, PIPE_MAP_WRITE, &box, + upload, native_stride, native_stride); + fprintf(stderr, "tva-proc: native upload path=gpu stride=%u bytes=%02x %02x %02x %02x\n", + native_stride, upload[0], upload[1], upload[2], upload[3]); + free(upload); + } else if (pipe->texture_map && pipe->texture_unmap) { + struct pipe_transfer *transfer = NULL; + uint8_t *map = pipe->texture_map(pipe, native, 0, PIPE_MAP_WRITE, + &box, &transfer); + if (!map || !transfer) { + fprintf(stderr, "tva-proc: NATIVE_SAMPLE_TEST map failed\n"); + if (transfer) + pipe->texture_unmap(pipe, transfer); + pipe_resource_reference(&native, NULL); + return 1; + } + for (unsigned y = 0; y < native->height0; y++) { + uint8_t *row = map + (size_t)y * transfer->stride; + for (unsigned x = 0; x < native->width0; x++) { + row[x * native_blocksize + 0] = 0x00; + row[x * native_blocksize + 1] = 0x00; + row[x * native_blocksize + 2] = 0xff; + row[x * native_blocksize + 3] = 0xff; + } + } + fprintf(stderr, "tva-proc: native upload path=cpu stride=%u bytes=%02x %02x %02x %02x\n", + transfer->stride, map[0], map[1], map[2], map[3]); + pipe->texture_unmap(pipe, transfer); + } else { + pipe_resource_reference(&native, NULL); + return 1; + } + if (pipe->screen->resource_changed) + pipe->screen->resource_changed(pipe->screen, native); + + struct pipe_sampler_view sv_templ; + memset(&sv_templ, 0, sizeof(sv_templ)); + u_sampler_view_default_template(&sv_templ, native, native->format); + struct pipe_sampler_view *sv = pipe->create_sampler_view( + pipe, native, &sv_templ); + if (!sv) { + pipe_resource_reference(&native, NULL); + return 1; + } + + struct u_rect src_rect = {0, native->width0, 0, native->height0}; + struct u_rect dst_rect = {0, dst->width, 0, dst->height}; + vl_compositor_clear_layers(&proc->cstate); + vl_compositor_set_rgba_layer(&proc->cstate, &proc->compositor, 0, sv, + &src_rect, &dst_rect, NULL); + vl_compositor_set_layer_dst_area(&proc->cstate, 0, &dst_rect); + if (getenv("DMD_VA_OFFSCREEN_TEST")) { + /* KGSL can clear an imported linear target, but its 3D path may not + * rasterize directly into that external layout. Render into a + * driver-owned texture first, then exercise the copy path into the + * imported target. */ + struct pipe_resource off_template = *src_resources[0]; + off_template.format = PIPE_FORMAT_B8G8R8A8_UNORM; + off_template.bind = PIPE_BIND_RENDER_TARGET | PIPE_BIND_SAMPLER_VIEW; + off_template.usage = PIPE_USAGE_DEFAULT; + off_template.flags = 0; + off_template.next = NULL; + struct pipe_resource *offscreen = pipe->screen->resource_create( + pipe->screen, &off_template); + struct pipe_surface off_surface = {0}; + if (offscreen) + pipe_surface_init(pipe, &off_surface, offscreen, 0, 0); + fprintf(stderr, "tva-proc: offscreen target=%p surface=%p\n", + (void *)offscreen, (void *)offscreen ? (void *)&off_surface : NULL); + if (offscreen) { + vl_compositor_render(&proc->cstate, &proc->compositor, + &off_surface, NULL, false); + + if (pipe->texture_map && pipe->texture_unmap) { + struct pipe_box probe = { .x = 0, .y = 0, .z = 0, + .width = 8, .height = 1, .depth = 1 }; + struct pipe_transfer *transfer = NULL; + uint8_t *map = pipe->texture_map(pipe, offscreen, 0, + PIPE_MAP_READ, &probe, + &transfer); + if (map && transfer) { + fprintf(stderr, "tva-proc: offscreen probe stride=%u bytes=%02x %02x %02x %02x %02x %02x %02x %02x\n", + transfer->stride, map[0], map[1], map[2], map[3], + map[4], map[5], map[6], map[7]); + pipe->texture_unmap(pipe, transfer); + } else { + fprintf(stderr, "tva-proc: offscreen probe map failed\n"); + if (transfer) + pipe->texture_unmap(pipe, transfer); + } + } + + struct pipe_blit_info blit = {0}; + blit.src.resource = offscreen; + blit.src.level = 0; + blit.src.box.x = 0; + blit.src.box.y = 0; + blit.src.box.z = 0; + blit.src.box.width = (int)dst->width; + blit.src.box.height = (int)dst->height; + blit.src.box.depth = 1; + blit.src.format = offscreen->format; + blit.dst.resource = surfaces[0].texture; + blit.dst.level = surfaces[0].level; + blit.dst.box.x = 0; + blit.dst.box.y = 0; + blit.dst.box.z = 0; + blit.dst.box.width = (int)dst->width; + blit.dst.box.height = (int)dst->height; + blit.dst.box.depth = 1; + blit.dst.format = surfaces[0].format; + blit.mask = PIPE_MASK_RGBA; + blit.filter = PIPE_TEX_FILTER_NEAREST; + pipe->blit(pipe, &blit); + pipe_resource_reference(&off_surface.texture, NULL); + pipe_resource_reference(&offscreen, NULL); + } + } else { + vl_compositor_render(&proc->cstate, &proc->compositor, &surfaces[0], + NULL, false); + } + pipe->sampler_view_release(pipe, sv); + pipe_resource_reference(&native, NULL); + return 0; + } + + if (getenv("DMD_VA_Y_SAMPLE_TEST") && !dst_yuv) { + struct pipe_resource *src_resources[VL_NUM_COMPONENTS] = {0}; + src->get_resources(src, src_resources); + fprintf(stderr, "tva-proc: Y_SAMPLE_TEST src0=%p format=%s %ux%u\n", + (void *)src_resources[0], + src_resources[0] ? util_format_short_name(src_resources[0]->format) : "none", + src_resources[0] ? src_resources[0]->width0 : 0, + src_resources[0] ? src_resources[0]->height0 : 0); + if (!src_resources[0]) + return 1; + + struct pipe_sampler_view sv_templ; + memset(&sv_templ, 0, sizeof(sv_templ)); + u_sampler_view_default_template(&sv_templ, src_resources[0], + src_resources[0]->format); + struct pipe_sampler_view *sv = pipe->create_sampler_view( + pipe, src_resources[0], &sv_templ); + if (!sv) + return 1; + + struct u_rect src_rect = {0, src_resources[0]->width0, 0, + src_resources[0]->height0}; + struct u_rect dst_rect = {0, dst->width, 0, dst->height}; + vl_compositor_clear_layers(&proc->cstate); + vl_compositor_set_rgba_layer(&proc->cstate, &proc->compositor, 0, sv, + &src_rect, &dst_rect, NULL); + vl_compositor_set_layer_dst_area(&proc->cstate, 0, &dst_rect); + vl_compositor_render(&proc->cstate, &proc->compositor, &surfaces[0], + NULL, false); + pipe->sampler_view_release(pipe, sv); + return 0; + } + if (util_format_get_nr_components(src->buffer_format) == 1) { /* Identity */ vl_csc_get_rgbyuv_matrix(PIPE_VIDEO_VPP_MCF_RGB, src->buffer_format, dst->buffer_format, @@ -173,7 +463,14 @@ compositor_proc_process_frame(struct pipe_video_codec *codec, vl_compositor_set_buffer_layer(&proc->cstate, &proc->compositor, 0, src, ¶m->src_region, NULL, deinterlace); vl_compositor_set_layer_dst_area(&proc->cstate, 0, ¶m->dst_region); - vl_compositor_render(&proc->cstate, &proc->compositor, &surfaces[0], NULL, false); + if (getenv("DMD_VA_SOLID")) { + union pipe_color_union color = { .f = { 1.0f, 0.0f, 0.0f, 1.0f } }; + pipe->clear_render_target(pipe, &surfaces[0], &color, 0, 0, + dst->width, dst->height, false); + } else { + vl_compositor_render(&proc->cstate, &proc->compositor, + &surfaces[0], NULL, false); + } } return 0; @@ -181,13 +478,57 @@ compositor_proc_process_frame(struct pipe_video_codec *codec, static int compositor_proc_end_frame(struct pipe_video_codec *codec, - struct pipe_video_buffer *target, - struct pipe_picture_desc *picture) + struct pipe_video_buffer *target, + struct pipe_picture_desc *picture) { struct vl_compositor_proc *proc = (struct vl_compositor_proc *)codec; proc->b.context->flush(proc->b.context, picture->out_pipe_fence, picture->flush_flags); + if (getenv("DMD_VA_PROBE") && target && target->get_surfaces && + proc->b.context->texture_map && proc->b.context->texture_unmap) { + if (picture->out_pipe_fence && *picture->out_pipe_fence && + proc->b.context->screen->fence_finish) + proc->b.context->screen->fence_finish( + proc->b.context->screen, proc->b.context, + *picture->out_pipe_fence, OS_TIMEOUT_INFINITE); + struct pipe_surface *surfaces = target->get_surfaces(target); + struct pipe_resource *res = surfaces ? surfaces[0].texture : NULL; + if (res) { + struct pipe_box box = { .x = 0, .y = 0, .z = 0, + .width = 8, .height = 1, .depth = 1 }; + struct pipe_transfer *transfer = NULL; + uint8_t *map = proc->b.context->texture_map(proc->b.context, res, 0, + PIPE_MAP_READ, &box, + &transfer); + if (map && transfer) { + fprintf(stderr, "tva-proc: dst probe fmt=%s stride=%u bytes=%02x %02x %02x %02x %02x %02x %02x %02x\n", + util_format_short_name(res->format), transfer->stride, + map[0], map[1], map[2], map[3], map[4], map[5], map[6], map[7]); + proc->b.context->texture_unmap(proc->b.context, transfer); + } else { + fprintf(stderr, "tva-proc: dst probe map failed res=%p\n", (void *)res); + if (transfer) + proc->b.context->texture_unmap(proc->b.context, transfer); + } + + struct pipe_box center = { .x = 960, .y = 540, .z = 0, + .width = 1, .height = 1, .depth = 1 }; + transfer = NULL; + uint8_t *center_map = proc->b.context->texture_map(proc->b.context, + res, 0, + PIPE_MAP_READ, + ¢er, + &transfer); + if (center_map && transfer) { + fprintf(stderr, "tva-proc: dst center fmt=%s stride=%u bytes=%02x %02x %02x %02x\n", + util_format_short_name(res->format), transfer->stride, + center_map[0], center_map[1], center_map[2], center_map[3]); + proc->b.context->texture_unmap(proc->b.context, transfer); + } + } + } + return 0; } diff --git a/src/gallium/drivers/freedreno/a6xx/fd6_draw.cc b/src/gallium/drivers/freedreno/a6xx/fd6_draw.cc index f835f06a25ad..3ac287f03df9 100644 --- a/src/gallium/drivers/freedreno/a6xx/fd6_draw.cc +++ b/src/gallium/drivers/freedreno/a6xx/fd6_draw.cc @@ -8,6 +8,8 @@ */ #include "pipe/p_state.h" +#include +#include #include "util/u_memory.h" #include "util/u_prim.h" #include "util/u_string.h" @@ -324,6 +326,12 @@ draw_vbos(struct fd_context *ctx, const struct pipe_draw_info *info, struct fd6_context *fd6_ctx = fd6_context(ctx); struct fd6_emit emit; + if (getenv("DMD_VA_PROBE")) + fprintf(stderr, "tva-fd draw mode=%u prim=%u count=%u start=%u vs=%p fs=%p\n", + info->mode, ctx->screen->primtypes[info->mode], + draws ? draws[0].count : 0, draws ? draws[0].start : 0, + ctx->prog.vs, ctx->prog.fs); + emit.ctx = ctx; emit.info = info; emit.indirect = indirect; @@ -337,8 +345,11 @@ draw_vbos(struct fd_context *ctx, const struct pipe_draw_info *info, emit.prog = NULL; emit.draw_id = 0; - if (!(ctx->prog.vs && ctx->prog.fs)) + if (!(ctx->prog.vs && ctx->prog.fs)) { + if (getenv("DMD_VA_PROBE")) + fprintf(stderr, "tva-fd draw skipped: missing shader\n"); return; + } if (PIPELINE == HAS_TESS_GS) { if ((info->mode == MESA_PRIM_PATCHES) || ctx->prog.gs) { @@ -362,8 +373,11 @@ draw_vbos(struct fd_context *ctx, const struct pipe_draw_info *info, } /* bail if compile failed: */ - if (!emit.prog) + if (!emit.prog) { + if (getenv("DMD_VA_PROBE")) + fprintf(stderr, "tva-fd draw skipped: program lookup failed\n"); return; + } fixup_draw_state(ctx, &emit); diff --git a/src/gallium/drivers/freedreno/a6xx/fd6_texture.cc b/src/gallium/drivers/freedreno/a6xx/fd6_texture.cc index 3b589568e263..02a60d4123fe 100644 --- a/src/gallium/drivers/freedreno/a6xx/fd6_texture.cc +++ b/src/gallium/drivers/freedreno/a6xx/fd6_texture.cc @@ -14,7 +14,12 @@ #include "util/u_memory.h" #include "util/u_string.h" +#include +#include +#include + #include "freedreno_dev_info.h" +#include "fd6_barrier.h" #include "fd6_emit.h" #include "fd6_resource.h" #include "fd6_screen.h" @@ -561,6 +566,21 @@ fd6_sampler_view_update(struct fd_context *ctx, ctx->screen->info->props.has_z24uint_s8uint); memcpy(so->descriptor, view.descriptor, sizeof(so->descriptor)); } + + if (getenv("DMD_VA_PROBE") && + (format == PIPE_FORMAT_R8_UNORM || + format == PIPE_FORMAT_B8G8R8A8_UNORM)) { + fprintf(stderr, + "tva-fd sampler res=%p shared=%d fmt=%s bo=%u iova=%#llx " + "layout=%#x pitch=%u size=%llu desc=%08x,%08x,%08x,%08x,%08x,%08x,%08x,%08x\n", + (void *)prsc, rsc->b.is_shared, util_format_short_name(format), + fd_bo_handle(rsc->bo), (unsigned long long)fd_bo_get_iova(rsc->bo), + rsc->layout.slices[0].offset, rsc->layout.pitch0, + (unsigned long long)rsc->layout.size, + so->descriptor[0], so->descriptor[1], so->descriptor[2], + so->descriptor[3], so->descriptor[4], so->descriptor[5], + so->descriptor[6], so->descriptor[7]); + } } template @@ -817,6 +837,93 @@ fd6_texture_state(struct fd_context *ctx, mesa_shader_stage type) struct fd6_texture_state *state = NULL; struct fd6_texture_key key; + /* EGL/DRI dma-buf imports are shared with an external producer (the + * termux-va bridge). The producer can update the BO without touching this + * context's resource sequence number or submitting a Gallium batch. Make + * the consumer invalidate its texture cache immediately before the draw; + * this is deliberately keyed to shared resources so ordinary textures do + * not pay the extra barrier. */ + bool shared_texture = false; + for (unsigned i = 0; i < tex->num_textures; i++) { + if (tex->textures[i] && + fd_resource(tex->textures[i]->texture)->b.is_shared) { + shared_texture = true; + break; + } + } + if (getenv("DMD_VA_PROBE") && tex->num_textures) { + for (unsigned i = 0; i < tex->num_textures; i++) { + if (!tex->textures[i]) + continue; + struct fd_resource *rsc = + fd_resource(tex->textures[i]->texture); + fprintf(stderr, "tva-fd texture state stage=%d slot=%u res=%p " + "shared=%d bo=%u format=%s\n", type, i, + (void *)tex->textures[i]->texture, rsc->b.is_shared, + rsc->bo ? fd_bo_handle(rsc->bo) : 0, + util_format_short_name(tex->textures[i]->format)); + } + } + const char *bo_sync_env = getenv("DMD_VA_BO_SYNC"); + const bool bo_sync = !bo_sync_env || + (strcmp(bo_sync_env, "0") != 0 && + strcmp(bo_sync_env, "false") != 0 && + strcmp(bo_sync_env, "off") != 0); + const char *sync_every_env = getenv("DMD_VA_SYNC_EVERY_DRAW"); + const bool sync_every_draw = sync_every_env && + (strcmp(sync_every_env, "1") == 0 || + strcmp(sync_every_env, "true") == 0 || + strcmp(sync_every_env, "on") == 0); + bool external_barrier = false; + if (shared_texture || getenv("DMD_VA_SYNC_ALL") || sync_every_draw) { + for (unsigned i = 0; i < tex->num_textures; i++) { + if (!tex->textures[i]) + continue; + + struct fd_resource *rsc = + fd_resource(tex->textures[i]->texture); + if (!rsc->b.is_shared && !getenv("DMD_VA_SYNC_ALL") && + !sync_every_draw) + continue; + + fd_resource_lock(rsc); + const bool need_handoff = sync_every_draw || + !rsc->tva_external_sync_valid; + const bool need_barrier = need_handoff || + rsc->tva_external_barrier_pending; + if (need_barrier) + rsc->tva_external_barrier_pending = false; + fd_resource_unlock(rsc); + external_barrier |= need_barrier; + if (!need_handoff) + continue; + + int ret = bo_sync ? fd_bo_sync_to_gpu(rsc->bo) : 0; + if (ret == 0) { + fd_resource_lock(rsc); + rsc->tva_external_sync_valid = true; + fd_resource_unlock(rsc); + } else if (getenv("DMD_VA_LOG")) { + fprintf(stderr, "tva-fd GPUOBJ_SYNC failed res=%p bo=%p errno=%d\n", + (void *)rsc, (void *)rsc->bo, errno); + } + } + + if (external_barrier) { + const unsigned external_flushes = FD6_INVALIDATE_CACHE | + FD6_FLUSH_CACHE | + FD6_WAIT_MEM_WRITES; + if (ctx->batch) + ctx->batch->barrier |= external_flushes; + if (ctx->batch_nondraw) + ctx->batch_nondraw->barrier |= external_flushes; + if (getenv("DMD_VA_LOG")) + fprintf(stderr, "tva-fd shared texture stage=%d batch=%p barrier=%#x\n", + type, (void *)ctx->batch, + ctx->batch ? ctx->batch->barrier : 0); + } + } + if (unlikely(fd6_ctx->tex_cache_needs_invalidate)) handle_invalidates(ctx); @@ -906,6 +1013,19 @@ fd6_rebind_resource(struct fd_context *ctx, struct fd_resource *rsc) assert_dt struct fd6_context *fd6_ctx = fd6_context(ctx); + /* A dma-buf may have been written by another GPU/CPU context without a + * BO rebind. Rebuilding the sampler state is not enough: invalidate the + * consumer-side texture cache before the next draw that uses this + * resource. The barrier is attached to whichever batch is active; the + * normal state emission path will consume it before issuing the draw. */ + const unsigned external_flushes = FD6_INVALIDATE_CACHE | + FD6_FLUSH_CACHE | + FD6_WAIT_MEM_WRITES; + if (ctx->batch) + ctx->batch->barrier |= external_flushes; + if (ctx->batch_nondraw) + ctx->batch_nondraw->barrier |= external_flushes; + hash_table_foreach (fd6_ctx->tex_cache, entry) { struct fd6_texture_state *state = (struct fd6_texture_state *)entry->data; diff --git a/src/gallium/drivers/freedreno/freedreno_resource.c b/src/gallium/drivers/freedreno/freedreno_resource.c index bcb6c2c8e365..dcbfe05f8ff8 100644 --- a/src/gallium/drivers/freedreno/freedreno_resource.c +++ b/src/gallium/drivers/freedreno/freedreno_resource.c @@ -30,6 +30,12 @@ #include "freedreno_util.h" #include +#include +#include +#include +#include +#include +#include #include "drm-uapi/drm_fourcc.h" /* XXX this should go away, needed for 'struct winsys_handle' */ @@ -157,11 +163,65 @@ rebind_resource(struct fd_resource *rsc) assert_dt fd_screen_unlock(screen); } +/* + * External producers (for example a VA decoder writing a shared dma-buf) + * can update the backing BO without changing the BO handle or the resource + * sequence number. Gallium's external-image paths call resource_changed() + * in that situation so the driver can invalidate any derived texture state. + * Freedreno normally reaches this path when a resource is reallocated; make + * the same cache invalidation explicit for imported resources as well. + */ +static void +fd_resource_changed(struct pipe_screen *pscreen, struct pipe_resource *prsc) +{ + (void)pscreen; + if (!prsc) + return; + + if (getenv("DMD_VA_LOG")) + fprintf(stderr, "tva-fd resource_changed res=%p fmt=%d %ux%u\n", + (void *)prsc, prsc->format, prsc->width0, prsc->height0); + + struct fd_resource *rsc = fd_resource(prsc); + fd_resource_lock(rsc); + rsc->tva_external_sync_valid = false; + rsc->tva_external_barrier_pending = true; + fd_resource_unlock(rsc); + + /* The termux-va bridge writes a stable linear imported resource in place. + * Its sampler descriptor does not change between frames, so invalidate + * only the external cache handoff instead of rebuilding every shader + * state object on each frame. Keep the normal rebind path for all other + * external-image users. */ + const char *bridge_env = getenv("TERMUX_VA_BRIDGE"); + const bool tva_in_place = rsc->b.is_shared && bridge_env && + strcmp(bridge_env, "0") != 0 && + strcmp(bridge_env, "false") != 0 && + strcmp(bridge_env, "off") != 0; + if (!tva_in_place) { + fd_resource_set_usage(prsc, FD_DIRTY_TEX); + rebind_resource(rsc); + } + + /* A lowered multi-plane import is represented by a linked resource chain; + * invalidate each plane's cached texture state when the external producer + * publishes a new frame. */ + if (prsc->next) + fd_resource_changed(pscreen, prsc->next); +} + static inline void fd_resource_set_bo(struct fd_resource *rsc, struct fd_bo *bo) { struct fd_screen *screen = fd_screen(rsc->b.b.screen); + if (getenv("DMD_VA_PROBE") && + (rsc->b.is_shared || (bo && (bo->alloc_flags & FD_BO_SHARED)))) + fprintf(stderr, "tva-fd set_bo pid=%d res=%p old=%u new=%u shared=%d\n", + (int)getpid(), (void *)rsc, + rsc->bo ? rsc->bo->handle : 0, + bo ? bo->handle : 0, rsc->b.is_shared); + rsc->bo = bo; rsc->seqno = seqno_next_u16(&screen->rsc_seqno); } @@ -425,6 +485,13 @@ fd_try_shadow_resource(struct fd_context *ctx, struct fd_resource *rsc, */ struct fd_resource *shadow = fd_resource(pshadow); + if (getenv("DMD_VA_PROBE")) + fprintf(stderr, "tva-fd shadow pid=%d res=%p old=%u shadow=%p new=%u shared=%d/%d modifier=%" PRIx64 "\n", + (int)getpid(), (void *)rsc, + rsc->bo ? rsc->bo->handle : 0, (void *)shadow, + shadow->bo ? shadow->bo->handle : 0, + rsc->b.is_shared, shadow->b.is_shared, modifier); + DBG("shadow: %p (%d, %p) -> %p (%d, %p)", rsc, rsc->b.b.reference.count, rsc->track, shadow, shadow->b.b.reference.count, shadow->track); @@ -705,15 +772,36 @@ fd_resource_transfer_unmap(struct pipe_context *pctx, struct fd_resource *rsc = fd_resource(ptrans->resource); struct fd_transfer *trans = fd_transfer(ptrans); + if (getenv("DMD_VA_PROBE")) + fprintf(stderr, "tva-fd unmap res=%p bo=%u shared=%d usage=%#x staging=%p upload=%p\n", + (void *)ptrans->resource, rsc->bo ? rsc->bo->handle : 0, + rsc->b.is_shared, + ptrans->usage, (void *)trans->staging_prsc, + trans->upload_ptr); + if (trans->staging_prsc) { - if (ptrans->usage & PIPE_MAP_WRITE) + if (ptrans->usage & PIPE_MAP_WRITE) { + /* The CPU has just populated the staging BO. KGSL does not provide + * implicit cache maintenance for this CPU-to-GPU handoff, so clean + * it before the blit reads the staging contents. */ + fd_bo_sync_to_gpu(fd_resource(trans->staging_prsc)->bo); fd_blit_from_staging(ctx, trans); + } pipe_resource_reference(&trans->staging_prsc, NULL); } if (trans->upload_ptr) { fd_bo_upload(rsc->bo, trans->upload_ptr, ptrans->box.x, ptrans->box.width); + fd_bo_sync_to_gpu(rsc->bo); free(trans->upload_ptr); + } else if (ptrans->usage & PIPE_MAP_WRITE) { + /* Direct maps and upload-manager maps write the BO from the CPU. */ + fd_bo_sync_to_gpu(rsc->bo); + } else if (ptrans->usage & PIPE_MAP_READ) { + /* Complete the GPU-to-dma-buf transition after a shared resource has + * been observed by the CPU. This is needed by KGSL consumers that + * import the same dma-buf in a separate GPU context. */ + fd_bo_sync_to_gpu(rsc->bo); } util_range_add(&rsc->b.b, &rsc->valid_buffer_range, ptrans->box.x, @@ -950,6 +1038,15 @@ resource_transfer_map(struct pipe_context *pctx, struct pipe_resource *prsc, if (ret) return NULL; } + + /* A shared resource may have just been written by a GPU producer in a + * different API context (for example the VA compositor). KGSL does not + * provide an implicit dma-buf cache transition for that handoff. Flush + * the producer's cache before exposing the resource to a CPU mapping; + * this also makes the completed contents visible to a subsequent GPU + * import in another context. */ + if ((usage & PIPE_MAP_READ) && rsc->b.is_shared) + fd_bo_sync_to_gpu(rsc->bo); } return resource_transfer_map_unsync(pctx, prsc, level, usage, box, trans); @@ -1098,6 +1195,10 @@ fd_resource_get_handle(struct pipe_screen *pscreen, struct pipe_context *pctx, assert_dt { struct fd_resource *rsc = fd_resource(prsc); + /* Keep track of resources imported from an external handle. The + * is_shared bit is also set when a newly allocated resource is exported, + * so it must be sampled before this function marks the resource shared. */ + const bool imported = rsc->b.is_shared; rsc->b.is_shared = true; @@ -1110,16 +1211,19 @@ fd_resource_get_handle(struct pipe_screen *pscreen, struct pipe_context *pctx, handle->modifier = DRM_FORMAT_MOD_LINEAR; if (!(prsc->bind & PIPE_BIND_SHARED)) { - struct fd_context *ctx = fd_screen_aux_context_get(pscreen); - + /* Preserve the shared binding for callers which cache the resource + * usage, but never replace storage that was imported from an + * external dma-buf. */ prsc->bind |= PIPE_BIND_SHARED; - bool ret = fd_try_shadow_resource(ctx, rsc, 0, NULL, handle->modifier); - - fd_screen_aux_context_put(pscreen); - - if (!ret) - return false; + if (!imported) { + struct fd_context *ctx = fd_screen_aux_context_get(pscreen); + bool ret = fd_try_shadow_resource(ctx, rsc, 0, NULL, + handle->modifier); + fd_screen_aux_context_put(pscreen); + if (!ret) + return false; + } } } @@ -1134,8 +1238,10 @@ fd_resource_get_handle(struct pipe_screen *pscreen, struct pipe_context *pctx, bool ret = fd_screen_bo_get_handle(pscreen, rsc->bo, rsc->scanout, fd_resource_pitch(rsc, 0), handle); + if (ret) + handle->offset = fd_resource_offset(rsc, 0, handle->layer); - if (!ret && !(prsc->bind & PIPE_BIND_SHARED)) { + if (!ret && !imported && !(prsc->bind & PIPE_BIND_SHARED)) { pctx = threaded_context_unwrap_sync(pctx); @@ -1578,6 +1684,25 @@ fd_resource_from_handle(struct pipe_screen *pscreen, goto fail; } + if (getenv("DMD_VA_PROBE") && handle->type == WINSYS_HANDLE_TYPE_FD) { + struct stat st; + if (fstat((int)handle->handle, &st) == 0) + fprintf(stderr, "tva-fd import res=%p fd=%d dev=%ju ino=%ju bo=%u " + "fmt=%s %ux%u stride=%u offset=%u shared=%d\n", + (void *)rsc, (int)handle->handle, + (uintmax_t)st.st_dev, (uintmax_t)st.st_ino, + fd_bo_handle(bo), util_format_short_name(tmpl->format), + tmpl->width0, tmpl->height0, handle->stride, handle->offset, + rsc->b.is_shared); + else + fprintf(stderr, "tva-fd import res=%p fd=%d fstat errno=%d bo=%u " + "fmt=%s %ux%u stride=%u offset=%u shared=%d\n", + (void *)rsc, (int)handle->handle, errno, fd_bo_handle(bo), + util_format_short_name(tmpl->format), tmpl->width0, + tmpl->height0, handle->stride, handle->offset, + rsc->b.is_shared); + } + fd_resource_set_bo(rsc, bo); rsc->internal_format = tmpl->format; @@ -1845,6 +1970,7 @@ fd_resource_screen_init(struct pipe_screen *pscreen) pscreen->resource_from_handle = fd_resource_from_handle; pscreen->resource_get_handle = fd_resource_get_handle; pscreen->resource_get_param = fd_resource_get_param; + pscreen->resource_changed = fd_resource_changed; pscreen->resource_destroy = u_transfer_helper_resource_destroy; pscreen->transfer_helper = diff --git a/src/gallium/drivers/freedreno/freedreno_resource.h b/src/gallium/drivers/freedreno/freedreno_resource.h index db8768dccf18..1b070e6fe708 100644 --- a/src/gallium/drivers/freedreno/freedreno_resource.h +++ b/src/gallium/drivers/freedreno/freedreno_resource.h @@ -137,6 +137,11 @@ struct fd_resource { */ bool is_replacement : 1; + /* A dma-buf imported from an external producer needs one explicit cache + * handoff before the first GPU read after each resource_changed(). */ + bool tva_external_sync_valid : 1; + bool tva_external_barrier_pending : 1; + /* Uninitialized resources with UBWC format need their UBWC flag data * cleared before writes, as the UBWC state is read and used during * writes, so undefined UBWC flag data results in undefined results. diff --git a/src/gallium/drivers/freedreno/freedreno_screen.c b/src/gallium/drivers/freedreno/freedreno_screen.c index 1275dfb2e7dd..2026b503399c 100644 --- a/src/gallium/drivers/freedreno/freedreno_screen.c +++ b/src/gallium/drivers/freedreno/freedreno_screen.c @@ -6,6 +6,8 @@ * Rob Clark */ +#include "compiler/glsl_types.h" + #include "pipe/p_defines.h" #include "pipe/p_screen.h" #include "pipe/p_state.h" @@ -25,6 +27,7 @@ #include #include #include +#include #include "drm-uapi/drm_fourcc.h" #include "freedreno_fence.h" @@ -94,8 +97,14 @@ bool fd_binning_enabled = true; static bool fd_kgsl_dmabuf_enabled(void) { + /* The PRoot/container path has no DRM render node, so applications select + * the KGSL backend explicitly through TERMUX_VA_GPU_BACKEND. Honor that + * selection before the screen is created; enabling it from the VA bridge + * would be too late for ANGLE's GBM allocations. */ + const char *backend = getenv("TERMUX_VA_GPU_BACKEND"); return debug_get_bool_option("FD_KGSL_ENABLE_DMABUF", false) || - debug_get_bool_option("XWAYLAND_FORCE_KGSL_SURFACELESS", false); + debug_get_bool_option("XWAYLAND_FORCE_KGSL_SURFACELESS", false) || + (backend && strcmp(backend, "kgsl") == 0); } static const char * @@ -188,6 +197,8 @@ fd_screen_destroy(struct pipe_screen *pscreen) if (screen->compiler) ir3_screen_fini(pscreen); + glsl_type_singleton_decref(); + free(screen->perfcntr_queries); free(screen); } @@ -390,6 +401,13 @@ fd_init_screen_caps(struct fd_screen *screen) u_init_pipe_screen_caps(&screen->base, 1); + /* KGSL mappings are not CPU/GPU coherent on the Android kernels used by + * DRM-less containers. Do not let upload managers keep persistent maps: + * they otherwise never call buffer_unmap after writing vertex and + * constant data, leaving no point at which the cache can be cleaned. */ + if (screen->is_kgsl) + caps->buffer_map_persistent_coherent = false; + /* On the kgsl stack the screen's control fd may be a display/controller fd * rather than the kgsl GPU fd, so drmGetCap(DRM_CAP_PRIME) cannot describe * the backend's real import/export support. The kgsl backend allocates @@ -802,6 +820,7 @@ fd_screen_bo_get_handle(struct pipe_screen *pscreen, struct fd_bo *bo, struct fd_screen *screen = fd_screen(pscreen); whandle->stride = stride; + whandle->size = fd_bo_size(bo); if (whandle->type == WINSYS_HANDLE_TYPE_SHARED) { return fd_bo_get_name(bo, &whandle->handle) == 0; @@ -987,14 +1006,19 @@ fd_screen_create(int fd, if (!dev) return NULL; + glsl_type_singleton_init_or_ref(); + struct fd_screen *screen = CALLOC_STRUCT(fd_screen); struct pipe_screen *pscreen; uint64_t val; fd_screen_debug_init(); - if (!screen) + if (!screen) { + glsl_type_singleton_decref(); + fd_device_del(dev); return NULL; + } #ifdef HAVE_PERFETTO fd_perfetto_init(); diff --git a/src/gallium/frontends/dri/dri2.c b/src/gallium/frontends/dri/dri2.c index 292757a2ab6f..526e613dff9f 100644 --- a/src/gallium/frontends/dri/dri2.c +++ b/src/gallium/frontends/dri/dri2.c @@ -698,6 +698,11 @@ dri_create_image_from_winsys(struct dri_screen *screen, const unsigned format_planes = util_format_get_num_planes(map->pipe_format); uint64_t modifier = whandle[0].modifier; + if (getenv("DMD_VA_LOG")) + fprintf(stderr, "tva-dri import begin fourcc=%#x map=%s %ux%u handles=%d modifier=%#" PRIx64 "\n", + map->dri_fourcc, util_format_short_name(map->pipe_format), + width, height, num_handles, modifier); + if (format_and_modifier_supported(pscreen, map->pipe_format, screen->target, 0, 0, PIPE_BIND_RENDER_TARGET, modifier)) tex_usage |= PIPE_BIND_RENDER_TARGET; @@ -713,6 +718,17 @@ dri_create_image_from_winsys(struct dri_screen *screen, screen->target, 0, 0, PIPE_BIND_SAMPLER_VIEW, modifier)) { map = &r8_g8b8_mapping; tex_usage |= PIPE_BIND_SAMPLER_VIEW; + /* On the KGSL/ANGLE path, importing the high-level + * R8_G8B8_420 resource can produce stale or incorrectly sampled + * frames even though the underlying dma-buf planes are valid. Use the + * native R8/GR88 plane resources for that Android backend. */ + const char *backend = getenv("TERMUX_VA_GPU_BACKEND"); + if (backend && (!strcmp(backend, "kgsl") || + !strcmp(backend, "KGSL"))) + use_lowered = true; + if (getenv("DMD_VA_LOG")) + fprintf(stderr, "tva-dri import fallback NV12 -> %s\n", + util_format_short_name(map->pipe_format)); } /* For NV21, see if we have support for sampling r8_b8g8 */ @@ -833,6 +849,11 @@ dri_create_image_from_winsys(struct dri_screen *screen, if (!tex_usage) return NULL; + if (getenv("DMD_VA_LOG")) + fprintf(stderr, "tva-dri import selected map=%s usage=%#x lowered=%d planes=%u\n", + util_format_short_name(map->pipe_format), tex_usage, + use_lowered, format_planes); + img = CALLOC_STRUCT(dri_image); if (!img) return NULL; @@ -858,12 +879,20 @@ dri_create_image_from_winsys(struct dri_screen *screen, tex = pscreen->resource_from_handle(pscreen, &templ, &whandle[i], handle_usage); if (!tex) { + if (getenv("DMD_VA_LOG")) + fprintf(stderr, "tva-dri import resource failed extra plane=%d fmt=%s %ux%u stride=%u offset=%u\n", + i, util_format_short_name(templ.format), templ.width0, + templ.height0, whandle[i].stride, whandle[i].offset); pipe_resource_reference(&img->texture, NULL); FREE(img); return NULL; } img->texture = tex; + if (getenv("DMD_VA_LOG")) + fprintf(stderr, "tva-dri import resource extra plane=%d fmt=%s %ux%u stride=%u offset=%u ok\n", + i, util_format_short_name(templ.format), templ.width0, + templ.height0, whandle[i].stride, whandle[i].offset); } for (i = (use_lowered ? map->nplanes : format_planes) - 1; i >= 0; i--) { @@ -882,6 +911,11 @@ dri_create_image_from_winsys(struct dri_screen *screen, &templ, &whandle[use_lowered ? map->planes[i].buffer_index : i], handle_usage); if (!tex) { + if (getenv("DMD_VA_LOG")) + fprintf(stderr, "tva-dri import resource failed plane=%d fmt=%s %ux%u stride=%u offset=%u\n", + i, util_format_short_name(templ.format), templ.width0, + templ.height0, whandle[use_lowered ? map->planes[i].buffer_index : i].stride, + whandle[use_lowered ? map->planes[i].buffer_index : i].offset); pipe_resource_reference(&img->texture, NULL); FREE(img); return NULL; @@ -900,6 +934,11 @@ dri_create_image_from_winsys(struct dri_screen *screen, } img->texture = tex; + if (getenv("DMD_VA_LOG")) + fprintf(stderr, "tva-dri import resource plane=%d fmt=%s %ux%u stride=%u offset=%u ok\n", + i, util_format_short_name(templ.format), templ.width0, + templ.height0, whandle[use_lowered ? map->planes[i].buffer_index : i].stride, + whandle[use_lowered ? map->planes[i].buffer_index : i].offset); } img->level = 0; @@ -958,9 +997,24 @@ dri_create_image(struct dri_screen *screen, struct pipe_resource templ; unsigned tex_usage = 0; unsigned count = _count; + int image_format = format; - if (!map) + /* Planar GBM formats use their Gallium alias as the create-image input, + * while the DRI mapping table only carries the DRM FourCC. */ + if (!map && format == PIPE_FORMAT_R8_G8B8_420_UNORM) { + map = &r8_g8b8_mapping; + image_format = PIPE_FORMAT_NV12; + } + + if (getenv("DMD_VA_LOG") && format == PIPE_FORMAT_R8_G8B8_420_UNORM) + fprintf(stderr, "tva-dri create image format=%d map=%p size=%dx%d\n", format, + (void *) map, width, height); + + if (!map) { + if (getenv("DMD_VA_LOG")) + fprintf(stderr, "tva-dri create image has no format mapping\\n"); return NULL; + } if (!pscreen->resource_create_with_modifiers && count > 0) return NULL; @@ -978,8 +1032,12 @@ dri_create_image(struct dri_screen *screen, PIPE_BIND_SAMPLER_VIEW | PIPE_BIND_SAMPLER_VIEW_SUBOPTIMAL)) tex_usage |= PIPE_BIND_SAMPLER_VIEW; - if (!tex_usage) + if (!tex_usage) { + if (getenv("DMD_VA_LOG") && format == PIPE_FORMAT_R8_G8B8_420_UNORM) + fprintf(stderr, "tva-dri create image format unsupported pipe=%u\n", + map->pipe_format); return NULL; + } if (use & __DRI_IMAGE_USE_SCANOUT) tex_usage |= PIPE_BIND_SCANOUT; @@ -1011,24 +1069,66 @@ dri_create_image(struct dri_screen *screen, templ.depth0 = 1; templ.array_size = 1; - if (modifiers) + if (map->nplanes > 1) { + struct pipe_resource *next = NULL; + + /* Planar GBM allocations are represented by one resource per plane. + * The resources may use separate dma-bufs; GBM exposes them through + * the per-plane handle accessors. */ + for (int plane = map->nplanes - 1; plane >= 0; plane--) { + struct pipe_resource plane_templ = templ; + + if (map->planes[plane].dri_format == __DRI_IMAGE_FORMAT_NONE) { + pipe_resource_reference(&next, NULL); + FREE(img); + return NULL; + } + + plane_templ.format = map->planes[plane].dri_format; + plane_templ.width0 = width >> map->planes[plane].width_shift; + plane_templ.height0 = height >> map->planes[plane].height_shift; + plane_templ.next = next; + + struct pipe_resource *resource; + if (modifiers) + resource = pscreen->resource_create_with_modifiers(pscreen, + &plane_templ, + modifiers, + count); + else + resource = pscreen->resource_create(pscreen, &plane_templ); + + if (!resource) { + pipe_resource_reference(&next, NULL); + FREE(img); + return NULL; + } + + next = resource; + } + img->texture = next; + } else if (modifiers) { img->texture = screen->base.screen ->resource_create_with_modifiers(screen->base.screen, &templ, modifiers, count); - else + } else { img->texture = screen->base.screen->resource_create(screen->base.screen, &templ); + } if (!img->texture) { + if (getenv("DMD_VA_LOG") && format == PIPE_FORMAT_R8_G8B8_420_UNORM) + fprintf(stderr, "tva-dri create image resource allocation failed pipe=%u\n", + map->pipe_format); FREE(img); return NULL; } img->level = 0; img->layer = 0; - img->dri_format = format; + img->dri_format = image_format; img->dri_fourcc = map->dri_fourcc; img->use = use; img->in_fence_fd = -1; @@ -1419,6 +1519,11 @@ dri2_from_dma_bufs(struct dri_screen *screen, struct dri_image *img; const struct dri2_format_mapping *map = dri2_get_mapping_by_fourcc(fourcc); + if (getenv("DMD_VA_LOG")) + fprintf(stderr, "tva-dri from_dma_bufs fourcc=%#x %ux%u modifier=%#" PRIx64 " fds=%d map=%s\n", + fourcc, width, height, modifier, num_fds, + map ? util_format_short_name(map->pipe_format) : "none"); + if (!screen->dmabuf_import) { if (error) *error = __DRI_IMAGE_ERROR_BAD_PARAMETER; @@ -1439,11 +1544,17 @@ dri2_from_dma_bufs(struct dri_screen *screen, const int expected_num_fds = dri2_get_modifier_num_planes(screen, modifier, fourcc); if (!map || expected_num_fds == 0) { + if (getenv("DMD_VA_LOG")) + fprintf(stderr, "tva-dri from_dma_bufs reject map=%p expected_fds=%d\n", + (void *)map, expected_num_fds); err = __DRI_IMAGE_ERROR_BAD_MATCH; goto exit; } if (num_fds != expected_num_fds) { + if (getenv("DMD_VA_LOG")) + fprintf(stderr, "tva-dri from_dma_bufs reject fd count=%d expected=%d\n", + num_fds, expected_num_fds); err = __DRI_IMAGE_ERROR_BAD_MATCH; goto exit; } diff --git a/src/gallium/frontends/dri/dri_helpers.c b/src/gallium/frontends/dri/dri_helpers.c index 68e9ae6395a8..c40ac74e6b23 100644 --- a/src/gallium/frontends/dri/dri_helpers.c +++ b/src/gallium/frontends/dri/dri_helpers.c @@ -21,6 +21,7 @@ */ #include +#include #include "drm-uapi/drm_fourcc.h" #include "util/u_memory.h" #include "pipe/p_screen.h" @@ -742,6 +743,14 @@ dri2_get_mapping_by_format(int format) return &dri2_format_table[i]; } + /* GBM passes the Gallium format for planar images because there is no + * legacy DRI image-format token for those formats. */ + for (unsigned i = 0; i < ARRAY_SIZE(dri2_format_table); i++) { + if (dri2_format_table[i].dri_format == __DRI_IMAGE_FORMAT_NONE && + dri2_format_table[i].pipe_format == format) + return &dri2_format_table[i]; + } + return NULL; } @@ -838,6 +847,12 @@ dri_create_image_with_modifiers(struct dri_screen *screen, unsigned int modifiers_count, void *loaderPrivate) { + if (getenv("DMD_VA_LOG")) + fprintf(stderr, "tva-dri create-with-modifiers format=%u use=%#x count=%u first=%#" PRIx64 " size=%ux%u\n", + dri_format, dri_usage, modifiers_count, + modifiers && modifiers_count ? modifiers[0] : 0, + width, height); + if (modifiers && modifiers_count > 0) { bool has_valid_modifier = false; int i; diff --git a/src/gallium/frontends/va/av1_bitstream.c b/src/gallium/frontends/va/av1_bitstream.c new file mode 100644 index 000000000000..e375b78747fa --- /dev/null +++ b/src/gallium/frontends/va/av1_bitstream.c @@ -0,0 +1,1240 @@ +/* + * SPDX-License-Identifier: GPL-3.0-or-later + * + * This file is a MODIFIED version of vaapi-driver/src/av1_bitstream.c from + * the droidspaces-media-decode project (Apache License 2.0), relicensed + * under GPL-3.0 for the Mesa termux-va bridge. + */ + +/* Reconstruct AV1 OBUs: variable-length codes, alignment and OBU headers. + * + * See the design notes at the top of av1_bitstream.h. This file implements + * the low-level bit-writing primitives together with sequence-header, + * frame-header and tile-group syntax. + */ +#include +#include +#include + +#include "av1_bitstream.h" + +/* ---------------------------------------------------------------- leb128 */ + +size_t dmd_av1_leb128_len(uint64_t v) +{ + size_t n = 0; + do { + n++; + v >>= 7; + } while (v && n < DMD_LEB128_MAX); + return n; +} + +size_t dmd_av1_leb128(uint64_t v, unsigned char *out, size_t out_cap) +{ + size_t n = 0; + do { + if (n >= out_cap || n >= DMD_LEB128_MAX) + return 0; + unsigned char byte = (unsigned char)(v & 0x7f); + v >>= 7; + if (v) + byte |= 0x80; /* another byte follows */ + out[n++] = byte; + } while (v); + return n; +} + +/* ------------------------------------------------------------------ uvlc */ + +void dmd_av1_put_uvlc(struct dmd_bitwriter *bw, uint32_t v) +{ + /* AV1 specification 4.10.3: write leadingZeros zero bits, one bit, and + * leadingZeros mantissa bits. The value v is encoded as + * (1 << leadingZeros) - 1 + mantissa. */ + uint32_t leading_zeros = 0; + uint64_t val = (uint64_t)v + 1; + + while ((val >> (leading_zeros + 1)) != 0) + leading_zeros++; + + if (leading_zeros >= 32) { + bw->overflow = 1; + return; + } + + /* Write leading_zeros zero bits; put_bits rejects nbits == 0, so skip the + * call for the empty prefix. */ + if (leading_zeros > 0) + dmd_bw_put_bits(bw, 0, (int)leading_zeros); + dmd_bw_put_flag(bw, 1); + if (leading_zeros > 0) + dmd_bw_put_bits(bw, (uint32_t)(val & ((1u << leading_zeros) - 1)), + (int)leading_zeros); +} + +/* -------------------------------------------------------------------- le */ + +void dmd_av1_put_le(struct dmd_bitwriter *bw, uint64_t v, int nbytes) +{ + /* AV1 specification 4.10.4: little-endian byte order; the caller must + * provide a byte-aligned bitstream. */ + if (nbytes <= 0 || nbytes > 8 || bw->bit_pos != 0) { + bw->overflow = 1; + return; + } + for (int i = 0; i < nbytes; i++) + dmd_bw_put_bits(bw, (uint32_t)((v >> (i * 8)) & 0xff), 8); +} + +/* -------------------------------------------------------------------- ns */ + +void dmd_av1_put_ns(struct dmd_bitwriter *bw, uint32_t v, uint32_t n) +{ + /* AV1 specification 4.10.7 ns(n): non-symmetric coding saves one bit for + * small values. + * w = FloorLog2(n) + 1 + * m = (1 << w) - n + * Write v directly in w - 1 bits when v < m; otherwise write v + m in + * w bits. */ + if (n == 0) { + bw->overflow = 1; + return; + } + if (n == 1) + return; /* the only value occupies no bits */ + + uint32_t w = 0, t = n; + while (t) { w++; t >>= 1; } /* w = FloorLog2(n) + 1 */ + uint32_t m = (1u << w) - n; + + if (v < m) { + dmd_bw_put_bits(bw, v, (int)(w - 1)); + } else { + uint32_t enc = v + m; + dmd_bw_put_bits(bw, enc >> 1, (int)(w - 1)); + dmd_bw_put_bits(bw, enc & 1, 1); + } +} + +/* -------------------------------------------------------------------- su */ + +void dmd_av1_put_su(struct dmd_bitwriter *bw, int32_t v, int nbits) +{ + /* AV1 specification 4.10.6 su(n): n-bit two's-complement coding. Keep + * the low nbits when writing; the decoder sign-extends the value. */ + if (nbits <= 0 || nbits > 32) { + bw->overflow = 1; + return; + } + uint32_t mask = (nbits == 32) ? 0xffffffffu : ((1u << nbits) - 1u); + dmd_bw_put_bits(bw, (uint32_t)v & mask, nbits); +} + +/* --------------------------------------------------------------- alignment */ + +void dmd_av1_byte_align(struct dmd_bitwriter *bw) +{ + /* AV1 specification 5.3.5 byte_alignment(): pad with zeros only; do not + * write a stop bit. This is the key difference from H.264/HEVC + * rbsp_trailing_bits. */ + while (bw->bit_pos != 0 && !bw->overflow) + dmd_bw_put_flag(bw, 0); +} + +void dmd_av1_trailing_bits(struct dmd_bitwriter *bw) +{ + /* AV1 specification 5.3.4 trailing_bits(): write one bit and then pad to + * a byte boundary. The one bit is required even when already aligned; it + * terminates the payload. */ + dmd_bw_put_flag(bw, 1); + while (bw->bit_pos != 0 && !bw->overflow) + dmd_bw_put_flag(bw, 0); +} + +/* ------------------------------------------------ sequence-header assembly */ + +/* Number of bits needed to represent x: AV1 uses FloorLog2(x) + 1. */ +static int bits_for(uint32_t v) +{ + int n = 0; + while (v) { n++; v >>= 1; } + return n ? n : 1; +} + +/* color_config(), AV1 specification 5.5.2. */ +static void put_color_config(struct dmd_bitwriter *bw, + const VADecPictureParameterBufferAV1 *p) +{ + const uint32_t depth_idx = p->bit_depth_idx; + const uint32_t mono = p->seq_info_fields.fields.mono_chrome; + const uint32_t sub_x = p->seq_info_fields.fields.subsampling_x; + const uint32_t sub_y = p->seq_info_fields.fields.subsampling_y; + + /* high_bitdepth / twelve_bit: bit_depth_idx 0/1/2 maps to 8/10/12 bits + * (va_dec_av1.h:255-260). Only profile 2 supports 12-bit output. */ + const int high_bitdepth = (depth_idx != 0); + dmd_bw_put_flag(bw, high_bitdepth); + if (p->profile == 2 && high_bitdepth) + dmd_bw_put_flag(bw, depth_idx == 2); /* twelve_bit */ + + if (p->profile != 1) + dmd_bw_put_flag(bw, (int)mono); /* mono_chrome */ + + /* color_description_present_flag = 0. + * + * We previously set this to 1 and wrote all three descriptors to retain + * VA-API's matrix_coefficients, but that did not match the real stream: + * libaom sets it to 0, leaving all three values UNSPECIFIED (2). Writing + * them would make the sequence header three bytes longer. + * + * This is acceptable because matrix_coefficients only selects the color + * conversion matrix; it does not affect decoding. The driver outputs + * NV12 and leaves color interpretation to the consumer. */ + dmd_bw_put_flag(bw, 0); + + if (mono) { + dmd_bw_put_flag(bw, (int)p->seq_info_fields.fields.color_range); + return; /* monochrome branch ends */ + } + + dmd_bw_put_flag(bw, (int)p->seq_info_fields.fields.color_range); + + /* Subsampling is explicit only for profile 2 at 12 bits. Profile 0 is + * always 4:2:0 and profile 1 is always 4:4:4, so the profile implies the + * values and no bits are present (specification 5.5.2). */ + if (p->profile == 2 && depth_idx == 2) { + dmd_bw_put_flag(bw, (int)sub_x); + if (sub_x) + dmd_bw_put_flag(bw, (int)sub_y); + } + + /* chroma_sample_position is present only for 4:2:0. The corresponding + * VA-API field is marked va_deprecated (:285), so use the value observed + * in the real stream (CSP_VERTICAL = 1). It only affects the assumed + * chroma-interpolation phase, not whether the frame can be decoded. */ + if (sub_x && sub_y) + dmd_bw_put_bits(bw, 1, 2); + + dmd_bw_put_flag(bw, 0); /* separate_uv_delta_q */ +} + +/* tile_info(), AV1 specification 5.9.15. + * + * tile_size_bytes_minus_1 is fixed at 3 (four bytes) here. tile_group must + * use the same width for tile_size_minus_1; otherwise the decoder reads tile + * lengths with the wrong width and becomes misaligned from the second tile. + * + * tile_info(), AV1 specification 5.9.15. + * + * The boundary calculations in this function were wrong twice; they were + * eventually checked line by line against FFmpeg's CBS implementation + * (tile_info() in libavcodec/cbs_av1_syntax_template.c). Two pitfalls are + * worth documenting: + * + * 1) The shift is sb_size = sb_shift + 2, not sb_shift. CBS says: + * sb_size = sb_shift + 2; + * max_tile_width_sb = AV1_MAX_TILE_WIDTH >> sb_size; + * Using sb_shift makes max_tile_width_sb four times too large, which + * makes min_log2_tile_cols too small and shifts the unary-code start. + * + * 2) min_log2_tiles must be at least min_log2_tile_cols. CBS says: + * min_log2_tiles = FFMAX(min_log2_tile_cols, + * cbs_av1_tile_log2(max_tile_area_sb, sb_rows*sb_cols)); + * Omitting this makes min_log2_tile_rows too small and emits extra ones + * in the row direction. + * + * increment(v, min, max) follows cbs_av1_write_increment: + * v == max -> write (max - min) ones, with no stop bit; + * otherwise -> write (v - min) ones followed by a zero. + * + * tile_size_bytes_minus_1 is fixed at 3 (four bytes), and tile_group must use + * the same width for tile_size_minus_1; a mismatch misaligns every tile after + * the first. */ +static void put_tile_info(struct dmd_bitwriter *bw, + const VADecPictureParameterBufferAV1 *p, + uint32_t mi_cols, uint32_t mi_rows) +{ + const int sb_shift = p->seq_info_fields.fields.use_128x128_superblock ? 5 : 4; + const int sb_size = sb_shift + 2; + + const uint32_t sb_cols = p->seq_info_fields.fields.use_128x128_superblock + ? ((mi_cols + 31) >> 5) : ((mi_cols + 15) >> 4); + const uint32_t sb_rows = p->seq_info_fields.fields.use_128x128_superblock + ? ((mi_rows + 31) >> 5) : ((mi_rows + 15) >> 4); + + const uint32_t MAX_TILE_COLS = 64, MAX_TILE_ROWS = 64; + const uint32_t max_tile_width_sb = 4096u >> sb_size; + const uint32_t max_tile_area_sb = (4096u * 2304u) >> (2 * sb_size); + + uint32_t max_log2_tile_cols = 0; + while ((1u << max_log2_tile_cols) < + (sb_cols < MAX_TILE_COLS ? sb_cols : MAX_TILE_COLS)) + max_log2_tile_cols++; + uint32_t max_log2_tile_rows = 0; + while ((1u << max_log2_tile_rows) < + (sb_rows < MAX_TILE_ROWS ? sb_rows : MAX_TILE_ROWS)) + max_log2_tile_rows++; + + uint32_t min_log2_tile_cols = 0; + while ((max_tile_width_sb << min_log2_tile_cols) < sb_cols) + min_log2_tile_cols++; + uint32_t min_log2_area = 0; + while ((max_tile_area_sb << min_log2_area) < sb_rows * sb_cols) + min_log2_area++; + const uint32_t min_log2_tiles = (min_log2_tile_cols > min_log2_area) + ? min_log2_tile_cols : min_log2_area; + + /* VA-API provides tile_cols/tile_rows as counts; derive their log2 values. */ + uint32_t cols_log2 = 0; + while ((1u << cols_log2) < p->tile_cols) + cols_log2++; + uint32_t rows_log2 = 0; + while ((1u << rows_log2) < p->tile_rows) + rows_log2++; + + + dmd_bw_put_flag(bw, (int)p->pic_info_fields.bits.uniform_tile_spacing_flag); + + if (p->pic_info_fields.bits.uniform_tile_spacing_flag) { + /* increment(tile_cols_log2, min_log2_tile_cols, max_log2_tile_cols). + * Clamp to the valid range first; CBS rejects out-of-range values on + * the write side. */ + if (cols_log2 < min_log2_tile_cols) cols_log2 = min_log2_tile_cols; + if (cols_log2 > max_log2_tile_cols) cols_log2 = max_log2_tile_cols; + for (uint32_t i = min_log2_tile_cols; i < cols_log2; i++) + dmd_bw_put_flag(bw, 1); + if (cols_log2 != max_log2_tile_cols) + dmd_bw_put_flag(bw, 0); + + const uint32_t min_log2_tile_rows = + (min_log2_tiles > cols_log2) ? (min_log2_tiles - cols_log2) : 0; + if (rows_log2 < min_log2_tile_rows) rows_log2 = min_log2_tile_rows; + if (rows_log2 > max_log2_tile_rows) rows_log2 = max_log2_tile_rows; + for (uint32_t i = min_log2_tile_rows; i < rows_log2; i++) + dmd_bw_put_flag(bw, 1); + if (rows_log2 != max_log2_tile_rows) + dmd_bw_put_flag(bw, 0); + } else { + /* Non-uniform spacing: write width_in_sbs_minus_1 and + * height_in_sbs_minus_1 for each tile using ns(n). The upper bound + * is the smaller of the remaining superblock count and max_tile_*_sb. */ + uint32_t start_sb = 0; + for (int i = 0; i < p->tile_cols && start_sb < sb_cols; i++) { + const uint32_t rest = sb_cols - start_sb; + const uint32_t lim = rest < max_tile_width_sb ? rest + : max_tile_width_sb; + dmd_av1_put_ns(bw, p->width_in_sbs_minus_1[i], lim); + start_sb += p->width_in_sbs_minus_1[i] + 1; + } + start_sb = 0; + for (int i = 0; i < p->tile_rows && start_sb < sb_rows; i++) { + dmd_av1_put_ns(bw, p->height_in_sbs_minus_1[i], + sb_rows - start_sb); + start_sb += p->height_in_sbs_minus_1[i] + 1; + } + } + + /* When TileCols * TileRows > 1, write context_update_tile_id and + * tile_size_bytes. Use the clamped log2 values above; do not recompute + * the width from tile_cols. */ + if (cols_log2 + rows_log2 > 0) { + dmd_bw_put_bits(bw, p->context_update_tile_id, + (int)(cols_log2 + rows_log2)); + /* tile_size_bytes_minus_1 = 1 (two bytes). The VA-API tile offsets + * have exactly two-byte gaps (observed as tile[0] offset 2 and each + * following tile starting two bytes after the previous end); those + * gaps are the original tile_size fields. Matching the source width + * keeps the reconstructed payload length unchanged and avoids + * needless expansion. This width must match the tile_size_minus_1 + * encoding in dmd_av1_build_frame(). */ + dmd_bw_put_bits(bw, 1, 2); + } +} + +/* quantization_params(), AV1 specification 5.9.12. */ +static void put_quantization_params(struct dmd_bitwriter *bw, + const VADecPictureParameterBufferAV1 *p) +{ + const uint32_t mono = p->seq_info_fields.fields.mono_chrome; + + dmd_bw_put_bits(bw, p->base_qindex, 8); + + /* delta_q uses su(1 + 6): one presence bit plus a six-bit signed value + * (specification 5.9.13, read_delta_q). A zero value writes only the + * presence bit set to zero. */ + #define PUT_DELTA_Q(v) do { \ + if ((v) != 0) { dmd_bw_put_flag(bw, 1); \ + dmd_av1_put_su(bw, (v), 7); } \ + else dmd_bw_put_flag(bw, 0); \ + } while (0) + + PUT_DELTA_Q(p->y_dc_delta_q); + + if (!mono) { + /* diff_uv_delta is present only when separate_uv_delta_q is set. The + * sequence header writes that flag as zero, so omit it here and use a + * shared U/V delta (writing the U value is sufficient). */ + PUT_DELTA_Q(p->u_dc_delta_q); + PUT_DELTA_Q(p->u_ac_delta_q); + } + + #undef PUT_DELTA_Q + + dmd_bw_put_flag(bw, (int)p->qmatrix_fields.bits.using_qmatrix); + if (p->qmatrix_fields.bits.using_qmatrix) { + dmd_bw_put_bits(bw, p->qmatrix_fields.bits.qm_y, 4); + dmd_bw_put_bits(bw, p->qmatrix_fields.bits.qm_u, 4); + if (!mono) + dmd_bw_put_bits(bw, p->qmatrix_fields.bits.qm_v, 4); + } +} + +/* segmentation_params(), AV1 specification 5.9.14. */ +static void put_segmentation_params(struct dmd_bitwriter *bw, + const VADecPictureParameterBufferAV1 *p, + int primary_ref_none) +{ + const uint32_t enabled = p->seg_info.segment_info_fields.bits.enabled; + dmd_bw_put_flag(bw, (int)enabled); + if (!enabled) + return; + + /* When primary_ref_frame is NONE, update_map/update_data are inferred as + * one and are not written. */ + if (!primary_ref_none) { + dmd_bw_put_flag(bw, (int)p->seg_info.segment_info_fields.bits.update_map); + if (p->seg_info.segment_info_fields.bits.update_map) + dmd_bw_put_flag(bw, + (int)p->seg_info.segment_info_fields.bits.temporal_update); + dmd_bw_put_flag(bw, (int)p->seg_info.segment_info_fields.bits.update_data); + } + + if (primary_ref_none || p->seg_info.segment_info_fields.bits.update_data) { + /* Write each segment and feature. Bits in feature_mask indicate + * enabled SEG_LVL_* features, while feature_data supplies the value. + * Width and signedness come from the Segmentation_Feature_Bits/Signed + * tables in the specification. */ + static const int seg_bits[8] = { 8, 6, 6, 6, 6, 3, 0, 0 }; + static const int seg_signed[8] = { 1, 1, 1, 1, 1, 0, 0, 0 }; + for (int i = 0; i < 8; i++) { + for (int j = 0; j < 8; j++) { + const int on = (p->seg_info.feature_mask[i] >> j) & 1; + dmd_bw_put_flag(bw, on); + if (!on) + continue; + if (seg_bits[j] == 0) + continue; /* SEG_LVL_REF_FRAME and other valueless features */ + if (seg_signed[j]) + dmd_av1_put_su(bw, p->seg_info.feature_data[i][j], + seg_bits[j] + 1); + else + dmd_bw_put_bits(bw, + (uint32_t)p->seg_info.feature_data[i][j], seg_bits[j]); + } + } + } +} + +/* loop_filter_params(), AV1 specification 5.9.11. */ +static void put_loop_filter_params(struct dmd_bitwriter *bw, + const VADecPictureParameterBufferAV1 *p, + int coded_lossless, int allow_intrabc) +{ + /* The entire section is omitted for lossless or allow_intrabc frames; + * the specification supplies the default values directly. */ + if (coded_lossless || allow_intrabc) + return; + + dmd_bw_put_bits(bw, p->filter_level[0], 6); + dmd_bw_put_bits(bw, p->filter_level[1], 6); + if (!p->seq_info_fields.fields.mono_chrome && + (p->filter_level[0] || p->filter_level[1])) { + dmd_bw_put_bits(bw, p->filter_level_u, 6); + dmd_bw_put_bits(bw, p->filter_level_v, 6); + } + dmd_bw_put_bits(bw, p->loop_filter_info_fields.bits.sharpness_level, 3); + + const int delta_enabled = + p->loop_filter_info_fields.bits.mode_ref_delta_enabled; + dmd_bw_put_flag(bw, delta_enabled); + if (delta_enabled) { + const int delta_update = + p->loop_filter_info_fields.bits.mode_ref_delta_update; + dmd_bw_put_flag(bw, delta_update); + if (delta_update) { + for (int i = 0; i < 8; i++) { + /* Write the update flag and su(7) for each entry. VA-API does + * not distinguish "not updated" from "updated to zero", so + * conservatively mark every non-zero value as an update. */ + if (p->ref_deltas[i] != 0) { + dmd_bw_put_flag(bw, 1); + dmd_av1_put_su(bw, p->ref_deltas[i], 7); + } else { + dmd_bw_put_flag(bw, 0); + } + } + for (int i = 0; i < 2; i++) { + if (p->mode_deltas[i] != 0) { + dmd_bw_put_flag(bw, 1); + dmd_av1_put_su(bw, p->mode_deltas[i], 7); + } else { + dmd_bw_put_flag(bw, 0); + } + } + } + } +} + +/* cdef_params(), AV1 specification 5.9.19. */ +static void put_cdef_params(struct dmd_bitwriter *bw, + const VADecPictureParameterBufferAV1 *p, + int coded_lossless, int allow_intrabc) +{ + if (coded_lossless || allow_intrabc || + !p->seq_info_fields.fields.enable_cdef) + return; + + dmd_bw_put_bits(bw, p->cdef_damping_minus_3, 2); + dmd_bw_put_bits(bw, p->cdef_bits, 2); + + const int n = 1 << p->cdef_bits; + for (int i = 0; i < n; i++) { + /* VA-API packs each Y/UV strength as (pri << 2) | sec, matching the + * four-bit pri plus two-bit sec layout in the bitstream. */ + dmd_bw_put_bits(bw, p->cdef_y_strengths[i] >> 2, 4); + dmd_bw_put_bits(bw, p->cdef_y_strengths[i] & 0x3, 2); + if (!p->seq_info_fields.fields.mono_chrome) { + dmd_bw_put_bits(bw, p->cdef_uv_strengths[i] >> 2, 4); + dmd_bw_put_bits(bw, p->cdef_uv_strengths[i] & 0x3, 2); + } + } +} + +/* lr_params(), AV1 specification 5.9.20. */ +static void put_lr_params(struct dmd_bitwriter *bw, + const VADecPictureParameterBufferAV1 *p, + int all_lossless, int allow_intrabc) +{ + if (all_lossless || allow_intrabc) + return; + + const uint32_t ry = p->loop_restoration_fields.bits.yframe_restoration_type; + const uint32_t rcb = p->loop_restoration_fields.bits.cbframe_restoration_type; + const uint32_t rcr = p->loop_restoration_fields.bits.crframe_restoration_type; + + /* The sequence header always sets enable_restoration to one, so this + * section is always present even when all three lr_type values are zero + * (the proper representation of no restoration for this frame). An + * earlier all-zero early return made the frame header six bits too short. */ + + /* lr_type uses f(2), in the same order as the VA-API restoration_type + * enum (0=NONE, 1=WIENER, 2=SGRPROJ, 3=SWITCHABLE). */ + dmd_bw_put_bits(bw, ry, 2); + if (!p->seq_info_fields.fields.mono_chrome) { + dmd_bw_put_bits(bw, rcb, 2); + dmd_bw_put_bits(bw, rcr, 2); + } + + if (ry || rcb || rcr) { + /* VA-API stores the decoded increment value (1 or 2) while the + * bitstream carries only the increment bit when 128x128 superblocks + * are enabled. */ + if (p->seq_info_fields.fields.use_128x128_superblock) { + const uint32_t shift = + p->loop_restoration_fields.bits.lr_unit_shift; + dmd_bw_put_bits(bw, shift > 0 ? shift - 1 : 0, 1); + } else { + const uint32_t shift = + p->loop_restoration_fields.bits.lr_unit_shift; + if (shift == 0) { + dmd_bw_put_bits(bw, 0, 1); + } else { + dmd_bw_put_bits(bw, 1, 1); + dmd_bw_put_bits(bw, shift > 1 ? 1 : 0, 1); + } + } + if (p->seq_info_fields.fields.use_128x128_superblock == 0 && + p->loop_restoration_fields.bits.lr_unit_shift) + dmd_bw_put_bits(bw, 0, 1); /* lr_unit_extra_shift */ + if (p->seq_info_fields.fields.subsampling_x && + p->seq_info_fields.fields.subsampling_y && (rcb || rcr)) + dmd_bw_put_bits(bw, p->loop_restoration_fields.bits.lr_uv_shift, 1); + } +} + +size_t dmd_av1_build_sequence_header(const void *pic_v, + unsigned char *out, size_t out_cap) +{ + const VADecPictureParameterBufferAV1 *p = pic_v; + if (!p || !out || out_cap < 8) + return 0; + + /* Write the payload to a temporary buffer first: obu_size uses leb128 and + * the payload length is needed before the header can be emitted. The + * sequence header is small (about 20 bytes in practice), so the stack + * buffer is sufficient. */ + unsigned char body[128]; + struct dmd_bitwriter bw; + dmd_bw_init(&bw, body, sizeof(body)); + + const uint32_t enable_order_hint = + p->seq_info_fields.fields.enable_order_hint; + + /* ---- sequence_header_obu(), AV1 specification 5.5.1 ---- */ + + dmd_bw_put_bits(&bw, p->profile, 3); /* seq_profile */ + dmd_bw_put_flag(&bw, (int)p->seq_info_fields.fields.still_picture); + dmd_bw_put_flag(&bw, 0); /* reduced_still_picture_header: always zero; + * setting it would omit most following fields, + * while frame parsing relies on enable_* bits. */ + + dmd_bw_put_flag(&bw, 0); /* timing_info_present_flag: VA-API does not + * provide timing_info. Timing does not affect + * decoding; the consumer controls presentation. */ + dmd_bw_put_flag(&bw, 0); /* initial_display_delay_present_flag */ + dmd_bw_put_bits(&bw, 0, 5);/* operating_points_cnt_minus_1: one operating + * point; this driver has no scalable layer, + * matching the OBU extension_flag of zero. */ + dmd_bw_put_bits(&bw, 0, 12); /* operating_point_idc[0] */ + /* seq_level_idx[0] = 9 (level 4.1) + seq_tier[0] = 0. + * + * The target stream is AV1 Main@L4.1; preserving level 4.1 avoids + * advertising a lower operating-point capability to strict decoders. + * The value is not the only detail: seq_level_idx > 7 must be followed by + * seq_tier (f(1)); omitting the zero bit shifts every subsequent field. + * VA-API does not provide the level, so use 8 to match the real stream. + * MediaCodec allocates from the actual resolution and does not validate + * this field. */ + dmd_bw_put_bits(&bw, 9, 5); /* seq_level_idx[0] */ + dmd_bw_put_flag(&bw, 0); /* seq_tier[0] (idx > 7) */ + + /* Use the minimum width needed for frame_width_bits rather than a fixed + * 16 bits. Per va_dec_av1.h:332-334, frame_width_minus1 is the upscaled + * frame size, which is the meaning required by max_frame_width_minus_1. */ + const uint32_t w_m1 = p->frame_width_minus1; + const uint32_t h_m1 = p->frame_height_minus1; + const int wbits = bits_for(w_m1); + const int hbits = bits_for(h_m1); + + dmd_bw_put_bits(&bw, (uint32_t)(wbits - 1), 4); + dmd_bw_put_bits(&bw, (uint32_t)(hbits - 1), 4); + dmd_bw_put_bits(&bw, w_m1, wbits); + dmd_bw_put_bits(&bw, h_m1, hbits); + + dmd_bw_put_flag(&bw, 0); /* frame_id_numbers_present_flag: zero because + * VA-API provides no delta_frame_id; the frame + * header must omit the matching fields too. */ + + dmd_bw_put_flag(&bw, (int)p->seq_info_fields.fields.use_128x128_superblock); + dmd_bw_put_flag(&bw, (int)p->seq_info_fields.fields.enable_filter_intra); + dmd_bw_put_flag(&bw, (int)p->seq_info_fields.fields.enable_intra_edge_filter); + dmd_bw_put_flag(&bw, (int)p->seq_info_fields.fields.enable_interintra_compound); + dmd_bw_put_flag(&bw, (int)p->seq_info_fields.fields.enable_masked_compound); + + /* VA-API does not provide the sequence-level enable_warped_motion field. + * Set the capability bit to one: the per-frame allow_warped_motion + * field (:439) still controls actual use. Setting it to zero while a + * frame requests warped motion would make the decoder reject that frame; + * setting it to one has no effect when the frame does not use it. */ + dmd_bw_put_flag(&bw, 1); + + dmd_bw_put_flag(&bw, (int)p->seq_info_fields.fields.enable_dual_filter); + dmd_bw_put_flag(&bw, (int)enable_order_hint); + if (enable_order_hint) { + dmd_bw_put_flag(&bw, (int)p->seq_info_fields.fields.enable_jnt_comp); + dmd_bw_put_flag(&bw, 1); /* enable_ref_frame_mvs: likewise unavailable + * from VA-API; use the safe value one (the + * per-frame field is use_ref_frame_mvs at :435). */ + } + + /* Match the sequence-control syntax used by the VA bitstream. When the + * current frame does not use screen-content tools, forcing both sequence + * controls to zero avoids inserting an extra frame-header bit. The + * entropy-coded tile payload is parsed with exactly this choice. */ + const int choose_screen_tools = + p->pic_info_fields.bits.allow_screen_content_tools; + dmd_bw_put_flag(&bw, choose_screen_tools); /* seq_choose_screen_content_tools */ + if (choose_screen_tools) { + /* seq_force_screen_content_tools is inferred as SELECT; the syntax + * therefore carries seq_choose_integer_mv. */ + dmd_bw_put_flag(&bw, 1); /* seq_choose_integer_mv */ + } else { + dmd_bw_put_flag(&bw, 0); /* seq_force_screen_content_tools */ + } + + if (enable_order_hint) + dmd_bw_put_bits(&bw, p->order_hint_bits_minus_1, 3); + + /* VA-API also omits enable_superres and enable_restoration, so infer them + * from the frame fields: + * use_superres (:432), and + * the three *frame_restoration_type values (:608-610; non-zero enables). + * Unlike warped_motion, these values change the frame-header syntax and + * must be accurate; a wrong value shifts the rest of the header. */ + const int use_superres = (int)p->pic_info_fields.bits.use_superres; + + dmd_bw_put_flag(&bw, use_superres); + dmd_bw_put_flag(&bw, (int)p->seq_info_fields.fields.enable_cdef); + /* enable_restoration is always one. + * + * Inferring this from whether all three frame_restoration_type values are + * zero was wrong. trace_headers shows enable_restoration=1 in the real + * stream, and the six lr_type[0..2] bits are present at frame-header bit + * 204 even though their values are all zero. In other words, a frame with + * no restoration is represented by lr_type=0, not by a sequence-level + * enable_restoration=0. The latter removes the lr_params section, makes + * the header six bits too short, and shifts tile_group parsing. + * + * One is the safe capability value; each frame's lr_type still determines + * whether restoration is actually used. */ + dmd_bw_put_flag(&bw, 1); + + put_color_config(&bw, p); + + dmd_bw_put_flag(&bw, + (int)p->seq_info_fields.fields.film_grain_params_present); + + dmd_av1_trailing_bits(&bw); + + if (bw.overflow) + return 0; + + /* Assemble the OBU header (including the leb128 payload length) and + * payload. */ + const size_t body_len = dmd_bw_bytes(&bw); + const size_t hdr = dmd_av1_obu_header(DMD_OBU_SEQUENCE_HEADER, + body_len, out, out_cap); + if (hdr == 0 || hdr + body_len > out_cap) + return 0; + for (size_t i = 0; i < body_len; i++) + out[hdr + i] = body[i]; + return hdr + body_len; +} + +/* Write uncompressed_header() to bw without trailing_bits or + * byte_alignment; the caller selects the terminator for the enclosing OBU: + * OBU_FRAME_HEADER (3) uses trailing_bits (specification 5.9.1); + * OBU_FRAME (6) uses byte_alignment (specification 5.10.1). + * This distinction was found through testing: the wrong terminator shifts + * the tile_group start and makes dav1d report "Failed to read unit". */ +static void put_uncompressed_header(struct dmd_bitwriter *bwp, + const VADecPictureParameterBufferAV1 *p, + uint8_t refresh_frame_flags) +{ + /* struct dmd_bitwriter is a value type: buf points to the caller's + * storage and the remaining fields are counters. Copying it in and back + * out is therefore safe, and preserves the field-by-field `&bw` writes + * validated against the real stream without introducing line-by-line + * changes. */ + struct dmd_bitwriter bw = *bwp; + + const uint32_t frame_type = p->pic_info_fields.bits.frame_type; + const int is_key = (frame_type == 0); /* KEY_FRAME */ + const int is_intra_only = (frame_type == 2); /* INTRA_ONLY_FRAME */ + const int intra_only = is_key || is_intra_only; + const uint32_t allow_intrabc = p->pic_info_fields.bits.allow_intrabc; + const uint32_t enable_order_hint = + p->seq_info_fields.fields.enable_order_hint; + const int order_hint_bits = enable_order_hint + ? (int)p->order_hint_bits_minus_1 + 1 : 0; + /* CodedLossless (specification 7.12.1) is true when all segment qindex + * values and the four delta_q values are zero. VA-API does not expose + * this flag directly, so derive it from the definition; it controls + * whether loop_filter, cdef and lr sections are present. */ + const int coded_lossless = + (p->base_qindex == 0 && p->y_dc_delta_q == 0 && + p->u_dc_delta_q == 0 && p->u_ac_delta_q == 0 && + p->v_dc_delta_q == 0 && p->v_ac_delta_q == 0); + /* AllLossless additionally requires no superres upscaling (specification + * 7.12.1). */ + const int all_lossless = + coded_lossless && !p->pic_info_fields.bits.use_superres; + + /* --- uncompressed_header(), AV1 specification 5.9.2 --- */ + + /* show_existing_frame: this driver forwards each decoded frame and does + * not reuse an existing frame, so the value is always zero. The field is + * still present when frame_id_numbers_present is zero; only current_frame_id + * is omitted in that case. */ + dmd_bw_put_flag(&bw, 0); + + dmd_bw_put_bits(&bw, frame_type, 2); + dmd_bw_put_flag(&bw, (int)p->pic_info_fields.bits.show_frame); + if (!p->pic_info_fields.bits.show_frame) + dmd_bw_put_flag(&bw, (int)p->pic_info_fields.bits.showable_frame); + + /* error_resilient_mode is inferred as one, and is not written, for a + * KEY_FRAME that is shown. */ + /* error_resilient_mode (from CBS): + * frame_type == SWITCH || (frame_type == KEY && show_frame) + * -> infer 1 (do not write it); + * otherwise write flag(error_resilient_mode). */ + const int er_inferred = (frame_type == 3) || + (is_key && p->pic_info_fields.bits.show_frame); + if (!er_inferred) + dmd_bw_put_flag(&bw, (int)p->pic_info_fields.bits.error_resilient_mode); + + dmd_bw_put_flag(&bw, (int)p->pic_info_fields.bits.disable_cdf_update); + + /* With sequence controls forced to zero, allow_screen_content_tools and + * force_integer_mv are inferred and do not occupy frame-header bits. If + * this frame requests the selectable mode, mirror the corresponding + * sequence-header form and emit the per-frame flags. */ + if (p->pic_info_fields.bits.allow_screen_content_tools) { + dmd_bw_put_flag(&bw, 1); /* allow_screen_content_tools */ + dmd_bw_put_flag(&bw, (int)p->pic_info_fields.bits.force_integer_mv); + } + + /* frame_id_numbers_present is zero in the sequence header, so skip + * current_frame_id. */ + + /* frame_size_override_flag is always zero: the frame size equals the + * sequence header's max_frame_size, which is set to this frame's size. + * A resolution change on a non-key frame recreates the session, so this + * invariant holds. */ + if (frame_type != 3 /* the flag is always one for SWITCH_FRAME */) + dmd_bw_put_flag(&bw, 0); + + /* order_hint (specification 5.9.2) is present when enable_order_hint is + * set and the condition "frame_is_intra and refresh_frame_flags == + * allFrames" is false. + * + * This condition caused two mistakes before it was checked against the + * real stream. A 1080p KEY_FRAME generated by libaom (show_frame=1) + * does write order_hint (value 0, after frame_size_override_flag), so it + * cannot be simplified to "never write it for intra frames"; doing so + * would omit order_hint_bits bits. + * + * For a KEY+show frame, refresh_frame_flags is absent from the stream; + * specification 7.20 derives RefreshFrameFlags = allFrames on the + * decoder side, while 5.9.2 applies to syntax variables. libaom's + * behavior confirms that the condition is false here, so order_hint is + * written. */ + if (enable_order_hint) + dmd_bw_put_bits(&bw, p->order_hint, order_hint_bits); + + /* primary_ref_frame is omitted for intra frames and error-resilient + * frames. */ + /* err_res is the effective value: it is one when inferred, otherwise it + * comes from VA-API. The conditions for primary_ref_frame, + * use_ref_frame_mvs and allow_warped_motion must use this value rather + * than the raw field. */ + const int err_res = er_inferred + ? 1 : (int)p->pic_info_fields.bits.error_resilient_mode; + const int primary_ref_none = intra_only || err_res || + (p->primary_ref_frame == 7 /* PRIMARY_REF_NONE */); + if (!intra_only && !err_res) + dmd_bw_put_bits(&bw, p->primary_ref_frame, 3); + + /* refresh_frame_flags is 0xFF and omitted from the stream for a KEY_FRAME + * with show_frame set. + * + * VA-API does not provide this field (a complete grep found only the + * reference in the comment at :421); it is the bitmask of reference slots + * refreshed by the frame. + * + * Use 0xFF (refresh all eight slots). MediaCodec manages reference-frame + * lifetime internally and does not allocate buffers from this bitmask; it + * only needs a syntactically valid value. Refreshing every slot is the + * conservative choice and avoids claiming that a slot remains valid after + * it has been overwritten. Reference management is less precise, but the + * driver forwards frames and lets MediaCodec reorder them. */ + const int refresh_all = (frame_type == 3) || + (is_key && p->pic_info_fields.bits.show_frame); + if (!refresh_all) + dmd_bw_put_bits(&bw, refresh_frame_flags, 8); + + /* Reference-frame indices are present only for inter frames. */ + if (!intra_only) { + /* frame_refs_short_signaling requires enable_order_hint. Set it to + * zero and write all seven ref_frame_idx values explicitly; VA-API + * supplies exactly this array. */ + if (enable_order_hint) + dmd_bw_put_flag(&bw, 0); + for (int i = 0; i < 7; i++) + dmd_bw_put_bits(&bw, p->ref_frame_idx[i], 3); + /* frame_id_numbers_present is zero, so do not write delta_frame_id. */ + } + + /* frame_size() / render_size(): with frame_size_override=0, the sequence + * header supplies the frame dimensions; write only superres and render_size + * (specifications 5.9.5/5.9.6/5.9.8). + * superres_params() (specification 5.9.8): use_superres is present in the + * stream only when the sequence header sets enable_superres. + * + * An earlier version always wrote this flag even though the sequence + * header had enable_superres = 0. The decoder then skipped the bit and + * the frame header was shifted by one, producing "trailing_one_bit out of + * range: 0" in dav1d. Whenever a sequence-level flag controls whether a + * frame field exists, both sides must use the same condition. */ + if (p->pic_info_fields.bits.use_superres) { + dmd_bw_put_flag(&bw, 1); + /* coded_denom: SUPERRES_DENOM_MIN = 9, using SUPERRES_DENOM_BITS = 3. */ + const uint32_t denom = p->superres_scale_denominator; + dmd_bw_put_bits(&bw, (denom >= 9 ? denom - 9 : 0), 3); + } + /* render_and_frame_size_different = 0: display size equals frame size. + * VA-API does not provide render_size; it affects display cropping only, + * not decoding. */ + dmd_bw_put_flag(&bw, 0); + + /* allow_intrabc (specification 5.9.2) is present when + * allow_screen_content_tools && UpscaledWidth == FrameWidth. FFmpeg's + * CBS implementation expresses the condition as: + * if (allow_screen_content_tools && upscaled_width == frame_width) + * flag(allow_intrabc); + * else + * infer(allow_intrabc, 0); + * + * upscaled_width == frame_width is equivalent to no superres upscaling: + * superres_params reduces frame_width by denom only when use_superres is + * set. + * + * Both conditions are required; omitting either adds or removes one bit. + * + * The CBS source and libaom's actual output differ here, so follow the + * real stream. + * + * CBS (cbs_av1_syntax_template.c) conditionally reads: + * if (allow_screen_content_tools && upscaled_width == frame_width) + * flag(allow_intrabc); else infer 0 + * + * A bit-level comparison with a libaom-generated 1080p KEY_FRAME showed + * allow_screen_content_tools=0. Skipping this bit as CBS does shifts the + * complete frame header and leaves trailing_one_bit=0; reading it + * unconditionally yields trailing_one_bit=1 and exact bit closure. + * + * Conclusion: the encoder writes this bit unconditionally. The criterion + * is whether the real stream closes correctly, not how the source code is + * written; the target is a decoder, not CBS. + * + * For allow_intrabc, trust ffmpeg trace_headers' bit-by-bit output for the + * real stream. In a libaom-generated 1080p KEY_FRAME with + * allow_screen_content_tools=0: + * bit 46 render_and_frame_size_different + * bit 47 disable_frame_end_update_cdf (allow_intrabc is absent) + * bit 48 uniform_tile_spacing_flag + * This confirms the CBS condition: when asct=0, the bit is omitted. + * + * An earlier conclusion based on bit closure in a home-grown parser was + * wrong: that parser was itself off by one here, and the two errors merely + * canceled out. Validation tools must first be calibrated against an + * authoritative source such as trace_headers. */ + if (intra_only && p->pic_info_fields.bits.allow_screen_content_tools && + !p->pic_info_fields.bits.use_superres) + dmd_bw_put_flag(&bw, (int)allow_intrabc); + + if (!intra_only) { + dmd_bw_put_flag(&bw, (int)p->pic_info_fields.bits.allow_high_precision_mv); + /* read_interpolation_filter(): is_filter_switchable f(1); when zero, + * write the two-bit interp_filter value. VA-API uses 4 for + * SWITCHABLE. */ + if (p->interp_filter == 4) { + dmd_bw_put_flag(&bw, 1); + } else { + dmd_bw_put_flag(&bw, 0); + dmd_bw_put_bits(&bw, p->interp_filter, 2); + } + dmd_bw_put_flag(&bw, + (int)p->pic_info_fields.bits.is_motion_mode_switchable); + /* use_ref_frame_mvs requires enable_ref_frame_mvs (set to one in the + * sequence header), a non-error-resilient frame and enable_order_hint. */ + if (!err_res && enable_order_hint) + dmd_bw_put_flag(&bw, (int)p->pic_info_fields.bits.use_ref_frame_mvs); + } + + /* disable_frame_end_update_cdf is present when + * reduced_still_picture_header=0 and disable_cdf_update=0. */ + if (!p->pic_info_fields.bits.disable_cdf_update) + dmd_bw_put_flag(&bw, + (int)p->pic_info_fields.bits.disable_frame_end_update_cdf); + + /* MiCols/MiRows: frame dimensions in 4x4 units (derived in specification + * 5.9.5). 2 * ((width + 7) >> 3) rounds up to eight pixels and halves the + * result. */ + const uint32_t width = (uint32_t)p->frame_width_minus1 + 1; + const uint32_t height = (uint32_t)p->frame_height_minus1 + 1; + const uint32_t mi_cols = 2 * ((width + 7) >> 3); + const uint32_t mi_rows = 2 * ((height + 7) >> 3); + + put_tile_info(&bw, p, mi_cols, mi_rows); + put_quantization_params(&bw, p); + put_segmentation_params(&bw, p, primary_ref_none); + + /* delta_q_params() (5.9.17): delta_q_present exists only when + * base_qindex > 0. */ + if (p->base_qindex > 0) + dmd_bw_put_flag(&bw, (int)p->mode_control_fields.bits.delta_q_present_flag); + if (p->mode_control_fields.bits.delta_q_present_flag) { + dmd_bw_put_bits(&bw, p->mode_control_fields.bits.log2_delta_q_res, 2); + + /* delta_lf_params() (5.9.18) is present only when delta_q_present is + * set. */ + if (!allow_intrabc) { + dmd_bw_put_flag(&bw, + (int)p->mode_control_fields.bits.delta_lf_present_flag); + if (p->mode_control_fields.bits.delta_lf_present_flag) { + dmd_bw_put_bits(&bw, + p->mode_control_fields.bits.log2_delta_lf_res, 2); + dmd_bw_put_flag(&bw, + (int)p->mode_control_fields.bits.delta_lf_multi); + } + } + } + + put_loop_filter_params(&bw, p, coded_lossless, (int)allow_intrabc); + put_cdef_params(&bw, p, coded_lossless, (int)allow_intrabc); + put_lr_params(&bw, p, all_lossless, (int)allow_intrabc); + + /* read_tx_mode() (5.9.21): CodedLossless implies ONLY_4X4 and writes no + * field; otherwise write tx_mode_select f(1). VA-API's tx_mode values are + * 2 = TX_MODE_LARGEST and 3 = TX_MODE_SELECT. */ + /* read_tx_mode (CBS source): + * coded_lossless -> infer(tx_mode, ONLY_4X4), no field; + * otherwise increment(tx_mode, TX_MODE_LARGEST=1, TX_MODE_SELECT=2). + * + * It was previously treated as an f(1) flag and encoded as tx_mode == 3; + * both assumptions were wrong: + * - the encoding is increment over the range [1,2], not a flag; + * - VA-API's tx_mode range is [0..2] (va_dec_av1.h:560-563), directly + * matching the specification enum, so 3 is not a valid value. + * increment writes max - min ones without a stop bit at range_max, and a + * single zero at range_min. */ + if (!coded_lossless) { + const uint32_t tm = p->mode_control_fields.bits.tx_mode; + if (tm >= 2) + dmd_bw_put_flag(&bw, 1); /* TX_MODE_SELECT: one, no stop bit */ + else + dmd_bw_put_flag(&bw, 0); /* TX_MODE_LARGEST: stop bit */ + } + + /* frame_reference_mode() (5.9.23): inter frames write reference_select. */ + if (!intra_only) + dmd_bw_put_flag(&bw, (int)p->mode_control_fields.bits.reference_select); + + /* skip_mode_params() (5.9.22): the VA descriptor carries the final + * skip_mode_present value, but not the reference order hints needed to + * derive skipModeAllowed. A stream with fewer than two populated + * reference slots cannot enable skip mode; this is also the conservative + * choice for descriptors whose reference indices are all the invalid + * first-frame value (7). For the normal inter frames, two or more + * populated slots are sufficient for the AV1 streams accepted here and + * the VA result can be emitted unchanged. + */ + unsigned populated_refs = 0; + if (!intra_only && p->mode_control_fields.bits.reference_select) { + for (int i = 0; i < 7; i++) + populated_refs += p->ref_frame_idx[i] != 7; + } + if (populated_refs >= 2) + dmd_bw_put_flag(&bw, (int)p->mode_control_fields.bits.skip_mode_present); + + /* allow_warped_motion requires is_motion_mode_switchable, a non-error- + * resilient frame and the sequence-level enable_warped_motion bit (set to + * one above). */ + if (!intra_only && + p->pic_info_fields.bits.is_motion_mode_switchable && !err_res) + dmd_bw_put_flag(&bw, (int)p->pic_info_fields.bits.allow_warped_motion); + + dmd_bw_put_flag(&bw, (int)p->mode_control_fields.bits.reduced_tx_set_used); + + /* global_motion_params() (5.9.24): write is_global for each reference on + * inter frames. VA-API supplies transform parameters in wm[], but + * encoding them back requires the complete differential coding and + * reference projection logic. Write is_global=0 (IDENTITY) for every + * reference instead; the known simplification loses global-motion + * compensation and is documented at the end of this file. */ + if (!intra_only) { + for (int i = 0; i < 7; i++) + dmd_bw_put_flag(&bw, 0); /* is_global[LAST+i] = 0 */ + } + + /* film_grain_params() (5.9.30): the section is absent when the sequence + * header's film_grain_params_present is zero. Mirror that condition here. */ + if (p->seq_info_fields.fields.film_grain_params_present && + (p->pic_info_fields.bits.show_frame || + p->pic_info_fields.bits.showable_frame)) + dmd_bw_put_flag(&bw, 0); /* apply_grain = 0 */ + + /* Do not write trailing_bits or byte_alignment; the caller selects the + * terminator for the enclosing OBU. */ + *bwp = bw; +} + +/* Wrap the frame-header payload in the requested OBU type. obu_type selects + * the terminator: + * DMD_OBU_FRAME_HEADER -> trailing_bits (specification 5.9.1); + * DMD_OBU_FRAME -> byte_alignment (specification 5.10.1), followed + * by tile_group. + * Return the number of bytes written to out, including the OBU header, or 0 + * on failure. body_len_out returns the payload length for OBU_FRAME, which + * appends tile_group afterwards. */ +static size_t build_frame_header_obu(const VADecPictureParameterBufferAV1 *p, + int obu_type, + unsigned char *body, size_t body_cap, + size_t *body_len_out, + uint8_t refresh_frame_flags) +{ + struct dmd_bitwriter bw; + dmd_bw_init(&bw, body, body_cap); + put_uncompressed_header(&bw, p, refresh_frame_flags); + + if (obu_type == DMD_OBU_FRAME) + dmd_av1_byte_align(&bw); /* zero padding only, no marker bit */ + else + dmd_av1_trailing_bits(&bw); /* one bit followed by zero padding */ + + if (bw.overflow) + return 0; + *body_len_out = dmd_bw_bytes(&bw); + return *body_len_out; +} + +size_t dmd_av1_build_frame_header(const void *pic_v, + unsigned char *out, size_t out_cap) +{ + const VADecPictureParameterBufferAV1 *p = pic_v; + if (!p || !out || out_cap < 8) + return 0; + + unsigned char body[512]; + size_t body_len = 0; + if (build_frame_header_obu(p, DMD_OBU_FRAME_HEADER, + body, sizeof(body), &body_len, 0xff) == 0) + return 0; + + const size_t hdr = dmd_av1_obu_header(DMD_OBU_FRAME_HEADER, + body_len, out, out_cap); + if (hdr == 0 || hdr + body_len > out_cap) + return 0; + for (size_t i = 0; i < body_len; i++) + out[hdr + i] = body[i]; + return hdr + body_len; +} + +size_t dmd_av1_build_frame(const void *pic_v, + const struct dmd_av1_tile *tiles, int num_tiles, + uint8_t refresh_frame_flags, + unsigned char *out, size_t out_cap) +{ + const VADecPictureParameterBufferAV1 *p = pic_v; + if (!p || !out || !tiles || num_tiles <= 0 || out_cap < 16) + return 0; + + /* Frame header (ends with byte_alignment, not trailing_bits). */ + unsigned char fh[512]; + size_t fh_len = 0; + if (build_frame_header_obu(p, DMD_OBU_FRAME, + fh, sizeof(fh), &fh_len, + refresh_frame_flags) == 0) + return 0; + /* tile_group_obu() payload (specification 5.11.1): + * when NumTiles > 1, write tile_start_and_end_present_flag first + * zero means that this group covers every tile (tg_start=0, + * tg_end=NumTiles-1) + * then byte_alignment, followed by tile_size_minus_1 and data per tile + * the final tile has no length field; its size is implied by the + * remaining OBU payload + * + * tile_size_minus_1 uses le(2), and its width must match the + * tile_size_bytes_minus_1 = 3 written by tile_info; a mismatch misaligns + * every tile after the first. */ + const uint32_t tile_total = + (uint32_t)p->tile_cols * (uint32_t)p->tile_rows; + + unsigned char tg_hdr[8]; + struct dmd_bitwriter tgw; + dmd_bw_init(&tgw, tg_hdr, sizeof(tg_hdr)); + if (tile_total > 1) + dmd_bw_put_flag(&tgw, 0); /* tile_start_and_end_present_flag */ + dmd_av1_byte_align(&tgw); + if (tgw.overflow) + return 0; + const size_t tg_hdr_len = dmd_bw_bytes(&tgw); + + /* Compute the total OBU payload length before writing the leb128 obu_size. */ + size_t payload_len = fh_len + tg_hdr_len; + for (int i = 0; i < num_tiles; i++) { + if (!tiles[i].data && tiles[i].len) + return 0; + payload_len += tiles[i].len; + if (i + 1 < num_tiles) + payload_len += 2; /* tile_size_minus_1, le(2) */ + } + + const size_t hdr = dmd_av1_obu_header(DMD_OBU_FRAME, payload_len, + out, out_cap); + if (hdr == 0 || hdr + payload_len > out_cap) + return 0; + + unsigned char *q = out + hdr; + for (size_t i = 0; i < fh_len; i++) + *q++ = fh[i]; + for (size_t i = 0; i < tg_hdr_len; i++) + *q++ = tg_hdr[i]; + for (int i = 0; i < num_tiles; i++) { + if (i + 1 < num_tiles) { + /* le(2): the width must equal tile_size_bytes_minus_1 + 1 = 2 + * from the frame header. This caps one tile at 64 KiB; observed + * 1080p tiles are about 4 KiB and the VA-API source stream also + * uses two bytes. If a future stream exceeds 64 KiB, widen this + * field and the frame-header field together. */ + const uint32_t v = (uint32_t)(tiles[i].len - 1); + *q++ = (unsigned char)(v & 0xFF); + *q++ = (unsigned char)((v >> 8) & 0xFF); + } + for (size_t k = 0; k < tiles[i].len; k++) + *q++ = tiles[i].data[k]; + } + return hdr + payload_len; +} + +size_t dmd_av1_build_show_existing(uint8_t map_idx, + unsigned char *out, size_t out_cap) +{ + if (!out || out_cap < 4 || map_idx >= 8) + return 0; + + unsigned char body[2]; + struct dmd_bitwriter bw; + dmd_bw_init(&bw, body, sizeof(body)); + dmd_bw_put_flag(&bw, 1); /* show_existing_frame */ + dmd_bw_put_bits(&bw, map_idx, 3); /* frame_to_show_map_idx */ + dmd_av1_trailing_bits(&bw); + if (bw.overflow) + return 0; + + const size_t body_len = dmd_bw_bytes(&bw); + const size_t hdr = dmd_av1_obu_header(DMD_OBU_FRAME_HEADER, + body_len, out, out_cap); + if (!hdr || hdr + body_len > out_cap) + return 0; + memcpy(out + hdr, body, body_len); + return hdr + body_len; +} + +/* --------------------------------------------------------------- OBU header */ + +size_t dmd_av1_obu_header(int obu_type, size_t payload_len, + unsigned char *out, size_t out_cap) +{ + if (obu_type < 0 || obu_type > 15 || out_cap < 1) + return 0; + + /* forbidden(1)=0 | type(4) | extension(1)=0 | has_size(1)=1 | reserved(1)=0 + * + * This is 0x00 | (type << 3) | 0x00 | 0x02 | 0x00; has_size is bit 1. + * Examples: SEQUENCE_HEADER (1) -> 0x0a, FRAME_HEADER (3) -> 0x1a, + * TILE_GROUP (4) -> 0x22, TEMPORAL_DELIMITER (2) -> 0x12. + * + * The invalid value observed before reconstruction was 0xd0 = + * 1101_0000: forbidden=1 (must be zero) and type=10 (reserved), clearly + * showing that the input was raw tile payload rather than an OBU header. */ + out[0] = (unsigned char)(((obu_type & 0x0f) << 3) | 0x02); + + size_t n = dmd_av1_leb128((uint64_t)payload_len, out + 1, out_cap - 1); + if (n == 0) + return 0; + return 1 + n; +} diff --git a/src/gallium/frontends/va/av1_bitstream.h b/src/gallium/frontends/va/av1_bitstream.h new file mode 100644 index 000000000000..50b17445611c --- /dev/null +++ b/src/gallium/frontends/va/av1_bitstream.h @@ -0,0 +1,180 @@ +/* + * SPDX-License-Identifier: GPL-3.0-or-later + * + * This file is a MODIFIED version of vaapi-driver/src/av1_bitstream.h from + * the droidspaces-media-decode project (Apache License 2.0), relicensed + * under GPL-3.0 for the Mesa termux-va bridge. + */ + +/* Reconstruct AV1 OBUs. + * + * VA-API passes AV1 decode parameters to the driver as structured fields + * (VADecPictureParameterBufferAV1), while the bitstream side provides only + * raw tile payloads with no OBU framing. The downstream MediaCodec decoder + * requires a complete AV1 OBU stream. + * + * libva documents this explicitly (/usr/include/va/va_dec_av1.h:643-645): + * "host decoder is responsible to parse out the per tile information. + * And the bit stream in sent to driver in per tile granularity." + * The same file (:637-638) also notes that VASliceParameterBufferAV1 + * "actually means VATileParameterBufferAV1". + * + * Warning: this is different from VP9 and must not be copied from that path. + * VP9 slice data is already a complete frame (va_dec_vp9.h:274-284), so no + * reconstruction is needed there. Applying the same approach to AV1 sends + * an invalid stream: the observed first byte is 0xd0 (forbidden_bit = 1 and + * reserved OBU type 10), and the decoder reports "No sequence header + * available" without decoding a frame. + * + * The bit-writing primitives are shared with bitstream.h: AV1 f(n) and the + * H.264/HEVC u(n) encodings are both fixed-width, MSB-first fields (AV1 + * specification 4.10.2). The AV1-specific variable-length encodings + * (leb128, uvlc and le) are implemented here. + */ +#ifndef DMD_AV1_BITSTREAM_H +#define DMD_AV1_BITSTREAM_H + +#include +#include + +#include "bitstream.h" + +/* OBU types (AV1 specification 6.2.2, obu_type table). */ +enum { + DMD_OBU_SEQUENCE_HEADER = 1, + DMD_OBU_TEMPORAL_DELIMITER = 2, + DMD_OBU_FRAME_HEADER = 3, + DMD_OBU_TILE_GROUP = 4, + DMD_OBU_METADATA = 5, + DMD_OBU_FRAME = 6, + DMD_OBU_REDUNDANT_FRAME_HEADER = 7, + DMD_OBU_TILE_LIST = 8, + DMD_OBU_PADDING = 15, +}; + +/* Maximum leb128 length (AV1 specification 4.10.5: eight bytes). */ +#define DMD_LEB128_MAX 8 + +/* --------------------------------------------------------- variable-length codes */ + +/* leb128(v): little-endian groups of seven bits, with the high bit of each + * byte indicating that another byte follows. Write to out and return the + * byte count, or 0 when out_cap is insufficient. */ +size_t dmd_av1_leb128(uint64_t v, unsigned char *out, size_t out_cap); + +/* Return the leb128 encoded length without writing; used to size obu_size. */ +size_t dmd_av1_leb128_len(uint64_t v); + +/* uvlc() (AV1 specification 4.10.3): a leading-zero count followed by a + * mantissa. Only a few frame-header fields use it; for example, + * timing_info's num_units_in_display_tick does not, while delta_frame_id + * does. */ +void dmd_av1_put_uvlc(struct dmd_bitwriter *bw, uint32_t v); + +/* le(n) (AV1 specification 4.10.4): an n-byte little-endian integer. + * The bitstream must be byte-aligned when this is called. */ +void dmd_av1_put_le(struct dmd_bitwriter *bw, uint64_t v, int nbytes); + +/* ns(n) (AV1 specification 4.10.7): non-symmetric binary coding. Used by + * tile_info for context_update_tile_id and tile-size derivation. */ +void dmd_av1_put_ns(struct dmd_bitwriter *bw, uint32_t v, uint32_t n); + +/* su(n) (AV1 specification 4.10.6): signed fixed-width coding, used by + * global_motion and related fields. */ +void dmd_av1_put_su(struct dmd_bitwriter *bw, int32_t v, int nbits); + +/* --------------------------------------------------------------- alignment */ + +/* byte_alignment() (AV1 specification 5.3.5): pad with zeros to a byte + * boundary. Unlike H.264 rbsp_trailing_bits, AV1 does not write a stop bit; + * using the wrong helper makes the decoder interpret padding as syntax. */ +void dmd_av1_byte_align(struct dmd_bitwriter *bw); + +/* trailing_bits() (AV1 specification 5.3.4): write one bit and then pad with + * zeros to a byte boundary. Used at the end of OBU payloads (sequence_header + * and frame_header, but not tile_group). */ +void dmd_av1_trailing_bits(struct dmd_bitwriter *bw); + +/* ---------------------------------------------------------------- OBU headers */ + +/* ------------------------------------------------ sequence-header assembly */ + +/* Build a complete OBU_SEQUENCE_HEADER from a + * VADecPictureParameterBufferAV1 (OBU header + payload + trailing_bits). + * + * pic is const void * rather than a concrete type: this header deliberately + * avoids including va_dec_av1.h, so libva dependencies do not spread to + * callers that only need the bitstream primitives. The implementation casts + * it to the concrete type. + * + * Write to out and return the total byte count; return 0 for insufficient + * capacity or invalid parameters. */ +size_t dmd_av1_build_sequence_header(const void *pic, + unsigned char *out, size_t out_cap); + +/* ---------------------------------------------- OBU_FRAME assembly (3/4 + 4/4) */ + +/* Tile offset and length used by dmd_av1_build_frame() to assemble a + * tile_group. */ +struct dmd_av1_tile { + const unsigned char *data; + size_t len; +}; + +/* Build a complete OBU_FRAME (6): frame header + byte_alignment + tile_group. + * + * Important: use OBU_FRAME rather than separate FRAME_HEADER (3) and + * TILE_GROUP (4) OBUs. dav1d rejects the split form with "Failed to read + * unit 0 (type 3)", while the combined form succeeds. Real libaom streams + * also use OBU_FRAME. + * + * The crucial detail is that the frame header inside OBU_FRAME ends with + * byte_alignment (zero padding), not trailing_bits (a one followed by + * padding). In specification 5.10.1 frame_obu(), byte_alignment() follows + * frame_header_obu(); using trailing_bits shifts the tile_group start and + * produces the same dav1d error. + * + * tiles are supplied in tile-row-major order, and the count must equal + * tile_cols * tile_rows. Write to out and return the total byte count; return + * 0 for insufficient capacity or invalid parameters. */ +size_t dmd_av1_build_frame(const void *pic, + const struct dmd_av1_tile *tiles, int num_tiles, + uint8_t refresh_frame_flags, + unsigned char *out, size_t out_cap); + +/* Build an OBU_FRAME_HEADER carrying show_existing_frame for a reference + * slot. The synthetic sequence header emitted by the bridge does not carry + * frame IDs, timing information, or a decoder model, so the payload consists + * only of show_existing_frame, frame_to_show_map_idx, and trailing_bits. */ +size_t dmd_av1_build_show_existing(uint8_t map_idx, + unsigned char *out, size_t out_cap); + +/* ---------------------------------------------------- frame-header assembly */ + +/* Build a complete OBU_FRAME_HEADER from a VADecPictureParameterBufferAV1 + * (OBU header + payload + trailing_bits). + * + * tile_cols and tile_rows come from pic. tile_size_bytes is fixed at four + * bytes and must match the tile_size_minus_1 width written in tile_group. + * + * Write to out and return the total byte count; return 0 for insufficient + * capacity or invalid parameters. */ +size_t dmd_av1_build_frame_header(const void *pic, + unsigned char *out, size_t out_cap); + +/* obu_header() + obu_size (AV1 specification 5.3.1/5.3.2). + * + * Bit layout (one byte without an extension): + * obu_forbidden_bit f(1) must be 0 + * obu_type f(4) + * obu_extension_flag f(1) always 0 (no scalable layer) + * obu_has_size_field f(1) always 1 (required by low-overhead format) + * obu_reserved_1bit f(1) must be 0 + * + * Write to out (one header byte followed by leb128 payload_len) and return the + * total byte count; return 0 when out_cap is insufficient. The caller then + * appends the payload at the returned offset. */ +size_t dmd_av1_obu_header(int obu_type, size_t payload_len, + unsigned char *out, size_t out_cap); + +#endif diff --git a/src/gallium/frontends/va/bitstream.c b/src/gallium/frontends/va/bitstream.c new file mode 100644 index 000000000000..f801a6b5674c --- /dev/null +++ b/src/gallium/frontends/va/bitstream.c @@ -0,0 +1,132 @@ +/* + * SPDX-License-Identifier: GPL-3.0-or-later + * + * This file is a MODIFIED version of vaapi-driver/src/bitstream.c from the + * droidspaces-media-decode project (Apache License 2.0), relicensed under + * GPL-3.0 for the Mesa termux-va bridge. + */ + +/* Shared H.264/HEVC bitstream-writing primitives; see bitstream.h. */ + +#include + +#include "bitstream.h" + + +void dmd_bw_init(struct dmd_bitwriter *bw, unsigned char *buf, size_t cap) +{ + bw->buf = buf; + bw->cap = cap; + bw->byte_pos = 0; + bw->bit_pos = 0; + bw->overflow = 0; + if (cap > 0) + buf[0] = 0; +} + +void dmd_bw_put_bits(struct dmd_bitwriter *bw, uint32_t value, int nbits) +{ + if (nbits <= 0 || nbits > 32) { + bw->overflow = 1; + return; + } + for (int i = nbits - 1; i >= 0; i--) { + if (bw->byte_pos >= bw->cap) { + bw->overflow = 1; + return; + } + unsigned int bit = (value >> i) & 1u; + bw->buf[bw->byte_pos] |= (unsigned char)(bit << (7 - bw->bit_pos)); + bw->bit_pos++; + if (bw->bit_pos == 8) { + bw->bit_pos = 0; + bw->byte_pos++; + if (bw->byte_pos < bw->cap) + bw->buf[bw->byte_pos] = 0; + } + } +} + +void dmd_bw_put_flag(struct dmd_bitwriter *bw, int v) +{ + dmd_bw_put_bits(bw, v ? 1u : 0u, 1); +} + +/* ue(v): unsigned Exp-Golomb code, written as (leadingZeros) 1 (info). + * The binary length of codeNum + 1 determines the number of leading zeros. */ +void dmd_bw_put_ue(struct dmd_bitwriter *bw, uint32_t v) +{ + if (v == 0xFFFFFFFFu) { /* v + 1 would wrap around. */ + bw->overflow = 1; + return; + } + uint32_t val = v + 1; + int nbits = 0; + while ((val >> nbits) != 0) + nbits++; + /* Write nbits - 1 leading zeros followed by the nbits bits of val (the + * most-significant 1 is the separator). When v == 0, nbits == 1 and + * there are no leading zeros; dmd_bw_put_bits(..., 0) would report an + * overflow because it rejects non-positive bit counts. */ + if (nbits > 1) + dmd_bw_put_bits(bw, 0, nbits - 1); + dmd_bw_put_bits(bw, val, nbits); +} + +/* se(v): signed Exp-Golomb code. Map 0, 1, -1, 2, -2, ... to + * 0, 1, 2, 3, 4, ... respectively. */ +void dmd_bw_put_se(struct dmd_bitwriter *bw, int32_t v) +{ + uint32_t code; + if (v <= 0) + code = (uint32_t)(-2 * (int64_t)v); + else + code = (uint32_t)(2 * (int64_t)v - 1); + dmd_bw_put_ue(bw, code); +} + +/* rbsp_trailing_bits: write one bit followed by zeros to the next byte + * boundary. */ +void dmd_bw_rbsp_trailing(struct dmd_bitwriter *bw) +{ + dmd_bw_put_flag(bw, 1); + while (bw->bit_pos != 0) + dmd_bw_put_flag(bw, 0); +} + +/* Number of bytes written. Call after rbsp_trailing, when the stream is + * byte-aligned. */ +size_t dmd_bw_bytes(const struct dmd_bitwriter *bw) +{ + return bw->bit_pos == 0 ? bw->byte_pos : bw->byte_pos + 1; +} + +/* ------------------------------------------------- emulation-prevention escaping */ + +/* Convert RBSP to SODB/EBSP by inserting 03 after every 00 00 0x sequence + * (x <= 3). MediaCodec receives complete NAL units in escaped form; + * otherwise a coincidental 00 00 01 sequence would be parsed as a start + * code. */ +size_t dmd_rbsp_escape(const unsigned char *rbsp, size_t len, + unsigned char *out, size_t out_cap) +{ + size_t o = 0; + int zeros = 0; + + for (size_t i = 0; i < len; i++) { + if (zeros >= 2 && rbsp[i] <= 0x03) { + if (o >= out_cap) + return 0; + out[o++] = 0x03; + zeros = 0; + } + if (o >= out_cap) + return 0; + out[o++] = rbsp[i]; + if (rbsp[i] == 0x00) + zeros++; + else + zeros = 0; + } + return o; +} diff --git a/src/gallium/frontends/va/bitstream.h b/src/gallium/frontends/va/bitstream.h new file mode 100644 index 000000000000..8922b49a3c76 --- /dev/null +++ b/src/gallium/frontends/va/bitstream.h @@ -0,0 +1,43 @@ +/* + * SPDX-License-Identifier: GPL-3.0-or-later + * + * This file is a MODIFIED version of vaapi-driver/src/bitstream.h from the + * droidspaces-media-decode project (Apache License 2.0), relicensed under + * GPL-3.0 for the Mesa termux-va bridge. + */ + +/* Shared H.264/HEVC bitstream-writing primitives. + * + * Both codecs use the same syntax-element encodings (u(n)/ue(v)/se(v), + * rbsp_trailing_bits and emulation prevention), so these helpers are shared + * instead of duplicating the H.264 implementation for HEVC. + */ +#ifndef DMD_BITSTREAM_H +#define DMD_BITSTREAM_H + +#include +#include + +struct dmd_bitwriter { + unsigned char *buf; + size_t cap; + size_t byte_pos; + int bit_pos; /* 0..7, number of bits written in the current byte */ + int overflow; /* Once set, writes are no-ops; callers check this. */ +}; + +void dmd_bw_init(struct dmd_bitwriter *bw, unsigned char *buf, size_t cap); +void dmd_bw_put_bits(struct dmd_bitwriter *bw, uint32_t value, int nbits); +void dmd_bw_put_flag(struct dmd_bitwriter *bw, int v); +void dmd_bw_put_ue(struct dmd_bitwriter *bw, uint32_t v); +void dmd_bw_put_se(struct dmd_bitwriter *bw, int32_t v); +void dmd_bw_rbsp_trailing(struct dmd_bitwriter *bw); +size_t dmd_bw_bytes(const struct dmd_bitwriter *bw); + +/* RBSP -> EBSP: insert 03 after every 00 00 0x sequence (x <= 3). + * MediaCodec requires the escaped representation. Return the number of + * bytes written to out, or 0 when the output buffer is too small. */ +size_t dmd_rbsp_escape(const unsigned char *rbsp, size_t len, + unsigned char *out, size_t out_cap); + +#endif diff --git a/src/gallium/frontends/va/buffer.c b/src/gallium/frontends/va/buffer.c index 53804f90a9b4..342a979c5acb 100644 --- a/src/gallium/frontends/va/buffer.c +++ b/src/gallium/frontends/va/buffer.c @@ -54,6 +54,9 @@ vlVaCreateBuffer(VADriverContextP ctx, VAContextID context, VABufferType type, vlVaDriver *drv; vlVaBuffer *buf; + fprintf(stderr, "tva-va: create buffer context=%u type=%d size=%u elements=%u\n", + context, type, size, num_elements); + if (!ctx) return VA_STATUS_ERROR_INVALID_CONTEXT; @@ -310,6 +313,7 @@ vlVaDestroyBuffer(VADriverContextP ctx, VABufferID buf_id) { vlVaDriver *drv; vlVaBuffer *buf; + vlVaContext *context; if (!ctx) return VA_STATUS_ERROR_INVALID_CONTEXT; @@ -322,6 +326,10 @@ vlVaDestroyBuffer(VADriverContextP ctx, VABufferID buf_id) return VA_STATUS_ERROR_INVALID_BUFFER; } + context = buf->ctx; + if (context) + mtx_lock(&context->mutex); + if (buf->derived_surface.resource) pipe_resource_reference(&buf->derived_surface.resource, NULL); @@ -336,12 +344,15 @@ vlVaDestroyBuffer(VADriverContextP ctx, VABufferID buf_id) FREE(buf->data); } - if (buf->ctx) { - assert(_mesa_set_search(buf->ctx->buffers, buf)); - _mesa_set_remove_key(buf->ctx->buffers, buf); + if (context) { + assert(_mesa_set_search(context->buffers, buf)); + _mesa_set_remove_key(context->buffers, buf); vlVaGetBufferFeedback(buf); - if (buf->fence && buf->ctx->decoder && buf->ctx->decoder->destroy_fence) - buf->ctx->decoder->destroy_fence(buf->ctx->decoder, buf->fence); + if (buf->fence && context->decoder && context->decoder->destroy_fence) { + context->decoder->destroy_fence(context->decoder, buf->fence); + buf->fence = NULL; + } + mtx_unlock(&context->mutex); } if (buf->coded_surf) diff --git a/src/gallium/frontends/va/config.c b/src/gallium/frontends/va/config.c index ca2a50d1fd79..a74cd924eff7 100644 --- a/src/gallium/frontends/va/config.c +++ b/src/gallium/frontends/va/config.c @@ -48,6 +48,8 @@ vlVaQueryConfigProfiles(VADriverContextP ctx, VAProfile *profile_list, int *num_ if (!ctx) return VA_STATUS_ERROR_INVALID_CONTEXT; + fprintf(stderr, "tva-va: query profiles\n"); + *num_profiles = 0; pscreen = VL_VA_PSCREEN(ctx); @@ -77,6 +79,8 @@ vlVaQueryConfigEntrypoints(VADriverContextP ctx, VAProfile profile, if (!ctx) return VA_STATUS_ERROR_INVALID_CONTEXT; + fprintf(stderr, "tva-va: query entrypoints profile=%d\n", profile); + *num_entrypoints = 0; if (profile == VAProfileNone) { @@ -620,6 +624,9 @@ vlVaCreateConfig(VADriverContextP ctx, VAProfile profile, VAEntrypoint entrypoin if (!ctx) return VA_STATUS_ERROR_INVALID_CONTEXT; + fprintf(stderr, "tva-va: create config profile=%d entrypoint=%d attrs=%d\n", + profile, entrypoint, num_attribs); + drv = VL_VA_DRIVER(ctx); pscreen = VL_VA_PSCREEN(ctx); @@ -792,6 +799,9 @@ vlVaQueryConfigAttributes(VADriverContextP ctx, VAConfigID config_id, VAProfile vlVaDriver *drv; vlVaConfig *config; + fprintf(stderr, "tva-va: query config attrs config=%u list=%p count=%d\n", + config_id, (void *)attrib_list, num_attribs ? *num_attribs : 0); + if (!ctx) return VA_STATUS_ERROR_INVALID_CONTEXT; @@ -829,5 +839,8 @@ vlVaQueryConfigAttributes(VADriverContextP ctx, VAConfigID config_id, VAProfile config->profile, config->entrypoint); + fprintf(stderr, "tva-va: query config attrs success profile=%d entrypoint=%d rt=%#x count=%d\n", + *profile, *entrypoint, attrib_list[0].value, *num_attribs); + return VA_STATUS_SUCCESS; } diff --git a/src/gallium/frontends/va/context.c b/src/gallium/frontends/va/context.c index a0f46fa75553..92e033313f26 100644 --- a/src/gallium/frontends/va/context.c +++ b/src/gallium/frontends/va/context.c @@ -41,6 +41,8 @@ #include "loader/loader.h" #endif +#include "tva_bridge.h" + #include static struct VADriverVTable vtable = @@ -133,6 +135,9 @@ VA_DRIVER_INIT_FUNC(VADriverContextP ctx) if (!ctx) return VA_STATUS_ERROR_INVALID_CONTEXT; + fprintf(stderr, "tva-va: driver init display_type=%lu\n", + (unsigned long)ctx->display_type); + drv = CALLOC(1, sizeof(vlVaDriver)); if (!drv) return VA_STATUS_ERROR_ALLOCATION_FAILED; @@ -160,6 +165,11 @@ VA_DRIVER_INIT_FUNC(VADriverContextP ctx) #endif if (!drv->vscreen) drv->vscreen = vl_dri3_screen_create(ctx->native_dpy, ctx->x11_screen); + /* PRoot exposes KGSL but no DRM render node. When the normal X11 + * DRI3 path cannot obtain a render device, let the termux-va bridge + * create its KGSL screen directly instead of falling back to swrast. */ + if (!drv->vscreen && tva_bridge_active()) + drv->vscreen = tva_bridge_vscreen_create(-1, false); if (!drv->vscreen) drv->vscreen = vl_xlib_swrast_screen_create(ctx->native_dpy, ctx->x11_screen); break; @@ -167,16 +177,22 @@ VA_DRIVER_INIT_FUNC(VADriverContextP ctx) case VA_DISPLAY_DRM: case VA_DISPLAY_DRM_RENDERNODES: { const struct drm_state *drm_info = (struct drm_state *) ctx->drm_state; + const int drm_fd = drm_info ? drm_info->fd : -1; - if (!drm_info || drm_info->fd < 0) { + /* A Wayland compositor in a PRoot container may expose no DRM fd at + * all. The termux-va bridge can open KGSL directly, so let it handle + * that case instead of rejecting the display before driver init. */ + if (drm_fd < 0 && !(tva_bridge_active() && + ctx->display_type == VA_DISPLAY_WAYLAND)) { FREE(drv); return VA_STATUS_ERROR_INVALID_PARAMETER; } #ifdef HAVE_DRISW_KMS - char* drm_driver_name = loader_get_driver_for_fd(drm_info->fd); + char* drm_driver_name = drm_fd >= 0 ? + loader_get_driver_for_fd(drm_fd) : NULL; if(drm_driver_name) { if (strcmp(drm_driver_name, "vgem") == 0) - drv->vscreen = vl_vgem_drm_screen_create(drm_info->fd); + drv->vscreen = vl_vgem_drm_screen_create(drm_fd); FREE(drm_driver_name); } #endif @@ -188,7 +204,10 @@ VA_DRIVER_INIT_FUNC(VADriverContextP ctx) * so don't try to override them. */ bool honor_dri_prime = ctx->display_type == VA_DISPLAY_WAYLAND; - drv->vscreen = vl_drm_screen_create(drm_info->fd, honor_dri_prime); + if (tva_bridge_active()) + drv->vscreen = tva_bridge_vscreen_create(drm_fd, honor_dri_prime); + else + drv->vscreen = vl_drm_screen_create(drm_fd, honor_dri_prime); } break; } @@ -201,19 +220,44 @@ VA_DRIVER_INIT_FUNC(VADriverContextP ctx) if (!drv->vscreen) goto error_screen; + fprintf(stderr, "tva-va: driver screen ready\n"); + + struct pipe_screen *raw_pscreen = drv->vscreen->pscreen; + + /* termux-va bridge: the underlying screen may lack the video capability + * hooks entirely (freedreno, llvmpipe); fill them in directly on the + * screen so the init check below passes and capability queries answer + * for the bridge's codec set. */ + if (tva_bridge_active()) + tva_bridge_screen_set_video_hooks(raw_pscreen); + /* video cannot work if these are not supported */ - if (!drv->vscreen->pscreen->get_video_param || !drv->vscreen->pscreen->is_video_format_supported) + if (!drv->vscreen->pscreen->get_video_param || !drv->vscreen->pscreen->is_video_format_supported) { + if (tva_bridge_active()) + fprintf(stderr, "tva: video capability hooks missing on the underlying screen\n"); goto error_pipe; + } bool compute_only = drv->vscreen->pscreen->caps.prefer_compute_for_multimedia; - drv->pipe = pipe_create_multimedia_context(drv->vscreen->pscreen, compute_only); - if (!drv->pipe) + drv->pipe = pipe_create_multimedia_context(raw_pscreen, compute_only); + if (!drv->pipe) { + if (tva_bridge_active()) + fprintf(stderr, "tva: multimedia context creation failed on the underlying screen\n"); goto error_pipe; + } + + if (tva_bridge_active()) + tva_bridge_pipe_set_codec_hooks(drv->pipe); drv->htab = handle_table_create(); if (!drv->htab) goto error_htab; + drv->surfaces = _mesa_set_create(NULL, _mesa_hash_pointer, + _mesa_key_pointer_equal); + if (!drv->surfaces) + goto error_surfaces; + (void) mtx_init(&drv->mutex, mtx_plain); ctx->pDriverData = (void *)drv; @@ -235,10 +279,18 @@ VA_DRIVER_INIT_FUNC(VADriverContextP ctx) snprintf(drv->vendor_string, sizeof(drv->vendor_string), "Mesa Gallium driver " PACKAGE_VERSION " for %s", drv->vscreen->pscreen->get_name(drv->vscreen->pscreen)); + if (tva_bridge_active()) { + size_t len = strlen(drv->vendor_string); + snprintf(drv->vendor_string + len, sizeof(drv->vendor_string) - len, + " (termux-va bridge)"); + } ctx->str_vendor = drv->vendor_string; return VA_STATUS_SUCCESS; +error_surfaces: + handle_table_destroy(drv->htab); + error_htab: drv->pipe->destroy(drv->pipe); @@ -265,13 +317,19 @@ vlVaCreateContext(VADriverContextP ctx, VAConfigID config_id, int picture_width, if (!ctx) return VA_STATUS_ERROR_INVALID_CONTEXT; + fprintf(stderr, "tva-va: create context config=%u size=%dx%d targets=%d\n", + config_id, picture_width, picture_height, num_render_targets); + drv = VL_VA_DRIVER(ctx); mtx_lock(&drv->mutex); config = handle_table_get(drv->htab, config_id); mtx_unlock(&drv->mutex); if (!config) + { + fprintf(stderr, "tva-va: create context invalid config\n"); return VA_STATUS_ERROR_INVALID_CONFIG; + } bool is_decode = config->entrypoint == PIPE_VIDEO_ENTRYPOINT_BITSTREAM; bool is_encode = config->entrypoint == PIPE_VIDEO_ENTRYPOINT_ENCODE; @@ -424,6 +482,8 @@ vlVaCreateContext(VADriverContextP ctx, VAConfigID config_id, int picture_width, } } + fprintf(stderr, "tva-va: create context success id=%u decoder=%p\n", + *context_id, (void *)context->decoder); return VA_STATUS_SUCCESS; } @@ -436,6 +496,8 @@ vlVaDestroyContext(VADriverContextP ctx, VAContextID context_id) if (!ctx) return VA_STATUS_ERROR_INVALID_CONTEXT; + fprintf(stderr, "tva-va: destroy context id=%u\n", context_id); + if (context_id == 0) return VA_STATUS_ERROR_INVALID_CONTEXT; @@ -554,6 +616,7 @@ vlVaTerminate(VADriverContextP ctx) drv->pipe2->destroy(drv->pipe2); drv->pipe->destroy(drv->pipe); drv->vscreen->destroy(drv->vscreen); + _mesa_set_destroy(drv->surfaces, NULL); handle_table_destroy(drv->htab); mtx_destroy(&drv->mutex); FREE(drv); diff --git a/src/gallium/frontends/va/meson.build b/src/gallium/frontends/va/meson.build index 86bfe2bcf5da..ca19eec5d5ad 100644 --- a/src/gallium/frontends/va/meson.build +++ b/src/gallium/frontends/va/meson.build @@ -12,6 +12,13 @@ libva_files = files( 'picture.c', 'surface.c', 'decode.c', 'image.c' ) +# termux-va bridge: forwards VA decode over a Unix socket to the Termux +# daemon. Compiled in unless disabled; activation is runtime-gated. +if with_termux_va_bridge + libva_files += files('tva_client.c', 'tva_bridge.c', 'tva_protocol.h', + 'bitstream.c', 'bitstream.h', 'av1_bitstream.c', 'av1_bitstream.h') +endif + if with_gfx_compute libva_files += files('postproc.c', 'subpicture.c') endif diff --git a/src/gallium/frontends/va/picture.c b/src/gallium/frontends/va/picture.c index a441ce471d6b..542810952924 100644 --- a/src/gallium/frontends/va/picture.c +++ b/src/gallium/frontends/va/picture.c @@ -37,6 +37,9 @@ #include "va_private.h" +#include +#include + void vlVaSetSurfaceContext(vlVaDriver *drv, vlVaSurface *surf, vlVaContext *context) { @@ -98,6 +101,9 @@ vlVaBeginPicture(VADriverContextP ctx, VAContextID context_id, VASurfaceID rende if (!drv) return VA_STATUS_ERROR_INVALID_CONTEXT; + fprintf(stderr, "tva-va: begin picture context=%u target=%u\n", + context_id, render_target); + mtx_lock(&drv->mutex); context = handle_table_get(drv->htab, context_id); if (!context) { @@ -194,6 +200,9 @@ vlVaRenderPicture(VADriverContextP ctx, VAContextID context_id, VABufferID *buff if (!drv) return VA_STATUS_ERROR_INVALID_CONTEXT; + fprintf(stderr, "tva-va: render picture context=%u buffers=%d\n", + context_id, num_buffers); + mtx_lock(&drv->mutex); context = handle_table_get(drv->htab, context_id); if (!context) { @@ -395,6 +404,11 @@ vlVaEndPicture(VADriverContextP ctx, VAContextID context_id) surf->coded_buf = coded_buf; } else if (context->decoder->entrypoint == PIPE_VIDEO_ENTRYPOINT_BITSTREAM) { context->desc.base.out_fence = &surf->fence; + if (getenv("DMD_VA_LOG")) + fprintf(stderr, "tva-picture end target=%#x surf=%p before_fence=%p out_ptr=%p entry=%d\n", + output_id, (void *)surf, (void *)surf->fence, + (void *)context->desc.base.out_fence, + context->decoder->entrypoint); } else if (context->decoder->entrypoint == PIPE_VIDEO_ENTRYPOINT_PROCESSING) { context->desc.base.out_fence = &surf->fence; context->desc.base.out_pipe_fence = &surf->pipe_fence; @@ -417,6 +431,11 @@ vlVaEndPicture(VADriverContextP ctx, VAContextID context_id) return VA_STATUS_ERROR_OPERATION_FAILED; } + if (getenv("DMD_VA_LOG")) + fprintf(stderr, "tva-picture end target=%#x surf=%p after_fence=%p out_ptr=%p\n", + output_id, (void *)surf, (void *)surf->fence, + (void *)context->desc.base.out_fence); + if (drv->pipe->screen->get_video_param(drv->pipe->screen, context->decoder->profile, context->decoder->entrypoint, @@ -426,6 +445,15 @@ vlVaEndPicture(VADriverContextP ctx, VAContextID context_id) if (context->decoder->entrypoint == PIPE_VIDEO_ENTRYPOINT_BITSTREAM) { if (context->proc.dst_surface) { + /* tva_codec_end_frame only queues the daemon request. The generic + * compositor must not sample the decode target until the bridge has + * copied that frame into its Gallium resources. */ + VAStatus sync_status = vlVaSyncSurfaceObjectLocked( + drv, surf, VA_TIMEOUT_INFINITE); + if (sync_status != VA_STATUS_SUCCESS) { + mtx_unlock(&drv->mutex); + return sync_status; + } if (!context->decoder->process_frame || context->decoder->process_frame(context->decoder, context->target, &context->proc.vpp) != 0) { VAStatus ret = diff --git a/src/gallium/frontends/va/postproc.c b/src/gallium/frontends/va/postproc.c index 1f68bf75581a..d2b9aa04ca0f 100644 --- a/src/gallium/frontends/va/postproc.c +++ b/src/gallium/frontends/va/postproc.c @@ -28,6 +28,8 @@ #include "util/u_handle_table.h" #include "util/u_memory.h" +#include + #include "vl/vl_defines.h" #include "vl/vl_video_buffer.h" #include "vl/vl_deint_filter.h" @@ -451,6 +453,63 @@ vlVaHandleVAProcPipelineParameterBufferType(vlVaDriver *drv, vlVaContext *contex if (!src || !dst) return VA_STATUS_ERROR_INVALID_SURFACE; + if (getenv("DMD_VA_PROBE")) + fprintf(stderr, "tva-proc: source surface=%#x surf=%p ctx=%p decoder=%p " + "fence=%p pipe=%p dst_surface=%p dst_ctx=%p\n", + param->surface, (void *)src_surface, (void *)src_surface->ctx, + src_surface->ctx ? (void *)src_surface->ctx->decoder : NULL, + (void *)src_surface->fence, (void *)src_surface->pipe_fence, + (void *)dst_surface, (void *)dst_surface->ctx); + + /* The termux-va decoder publishes frames asynchronously from a reader + * thread. Gallium's compositor does not consume pipe_vpp_desc::in_fence, + * so wait for the source surface before sampling it. Without this wait a + * decode+VPP submission can render the cleared (green) contents of a + * recycled NV12 surface. */ + /* PRIME imports do not carry the decoder's VA fence. Resolve the + * producer from the shared dma-buf just before sampling it, and wait only + * on that producer. This avoids blocking PRIME creation before Chromium + * has filled its decode pipeline. */ + vlVaSurface *producer = NULL; + uint64_t sync_timeout = VA_TIMEOUT_INFINITE; + /* The decoder and VPP calls are made on Chromium's single VA thread. + * Waiting indefinitely for a producer fence here can deadlock that same + * thread before it submits the next temporal unit. Keep the diagnostic + * no-wait mode consistent for both the public vaSyncSurface entry point + * and this internal VPP synchronization path. */ + if (vlVaSurfaceNoWait()) + sync_timeout = 0; + if (src_surface->is_prime_import) { + if (src_surface->sync_surface && + src_surface->prime_fence && + src_surface->sync_surface->fence == src_surface->prime_fence) { + producer = src_surface->sync_surface; + } else if (!src_surface->sync_surface) { + /* No producer was visible at import time. Retry only in that case; + * if the snapshot became stale, the dma-buf already belongs to an + * older frame and waiting on the replacement fence would deadlock. */ + producer = surface_find_prime_producer_for_surface(drv, src_surface); + /* A producer found only at sampling time may still be waiting for + * the decoder pipeline to accept more input. Do not block the + * application thread on that unverified generation: a later VPP + * submission will retry and copy it once the fence is ready. */ + if (producer) + sync_timeout = 0; + } + } + VAStatus sync_status = vlVaSyncSurfaceObjectLocked( + drv, producer ? producer : src_surface, sync_timeout); + if (sync_timeout == 0 && sync_status == VA_STATUS_ERROR_TIMEDOUT) + sync_status = VA_STATUS_SUCCESS; + if (getenv("DMD_VA_PROBE")) + fprintf(stderr, "tva-proc: source sync status=%d surf=%p producer=%p " + "ctx=%p fence=%p\n", + sync_status, (void *)src_surface, (void *)producer, + (void *)src_surface->ctx, + (void *)src_surface->fence); + if (sync_status != VA_STATUS_SUCCESS) + return sync_status; + for (i = 0; i < param->num_filters; i++) { vlVaBuffer *buf = handle_table_get(drv->htab, param->filters[i]); VAProcFilterParameterBufferBase *filter; diff --git a/src/gallium/frontends/va/surface.c b/src/gallium/frontends/va/surface.c index 5e78d7a9f5c5..492b26b8fea8 100644 --- a/src/gallium/frontends/va/surface.c +++ b/src/gallium/frontends/va/surface.c @@ -43,6 +43,132 @@ #include "va_private.h" +#include +#include +#include +#include +#ifndef _WIN32 +#include +#include +#include +#include +#include +#include "drm-uapi/dma-buf.h" +#endif + +static bool +vl_va_export_debug_enabled(void) +{ + const char *e = getenv("DMD_VA_LOG"); + return e && e[0] == '1'; +} + +static bool +vl_va_bridge_skip_clear(void) +{ + const char *e = getenv("TERMUX_VA_BRIDGE"); + return e && (e[0] == '1' || e[0] == 'y' || e[0] == 'Y'); +} + +static bool +vl_va_export_no_wait(void) +{ + const char *e = getenv("DMD_VA_EXPORT_NO_WAIT"); + return e && (e[0] == '1' || e[0] == 'y' || e[0] == 'Y'); +} + +bool +vlVaSurfaceNoWait(void) +{ + const char *e = getenv("DMD_VA_SURFACE_NO_WAIT"); + if (e && (e[0] == '1' || e[0] == 'y' || e[0] == 'Y')) + return true; + if (e && (e[0] == '0' || e[0] == 'n' || e[0] == 'N')) + return false; + +#ifndef _WIN32 + /* Chromium's PRoot VA path has no DRM render node. A blocking producer + * wait there prevents the same VA thread from submitting the next input + * unit, so use the bridge's asynchronous handoff. Chroot keeps the + * normal fence wait when a render node is available. */ + const char *bridge = getenv("TERMUX_VA_BRIDGE"); + const char *backend = getenv("TERMUX_VA_GPU_BACKEND"); + if (!bridge || !(bridge[0] == '1' || bridge[0] == 'y' || bridge[0] == 'Y') || + !backend || strcmp(backend, "kgsl") != 0 || + access("/dev/kgsl-3d0", R_OK) != 0) + return false; + + DIR *dir = opendir("/dev/dri"); + if (!dir) + return true; + + bool render_node = false; + struct dirent *entry; + while ((entry = readdir(dir))) { + if (strncmp(entry->d_name, "renderD", 7) == 0) { + render_node = true; + break; + } + } + closedir(dir); + return !render_node; +#else + return false; +#endif +} + +#define TVA_EXPORT_LOG(...) do { \ + if (vl_va_export_debug_enabled()) \ + fprintf(stderr, "tva-export: " __VA_ARGS__); \ +} while (0) + +#ifndef _WIN32 +static void +vl_va_export_probe_fd(int fd, uint64_t size, int plane) +{ + const char *e = getenv("DMD_VA_PROBE"); + if (!e || e[0] != '1' || fd < 0 || size < 8) + return; + + size_t map_size = size < 4096 ? (size_t)size : 4096; + bool sync_started = false; + struct dma_buf_sync sync = { + /* The bridge writes through a CPU mapping before exporting this fd. + * Use a write transaction here so cached lines are committed for a + * Vulkan/KGSL importer; READ-only sync does not promise that. */ + .flags = DMA_BUF_SYNC_START | DMA_BUF_SYNC_WRITE, + }; + if (ioctl(fd, DMA_BUF_IOCTL_SYNC, &sync) == 0) { + sync_started = true; + } else if (errno != ENOTTY && errno != EOPNOTSUPP && errno != ENOSYS) { + fprintf(stderr, "tva-export probe plane=%d sync start failed errno=%d\n", + plane, errno); + return; + } + + void *map = mmap(NULL, map_size, PROT_READ, MAP_SHARED, fd, 0); + if (map == MAP_FAILED) { + fprintf(stderr, "tva-export probe plane=%d mmap failed errno=%d\n", + plane, errno); + if (sync_started) { + sync.flags = DMA_BUF_SYNC_END | DMA_BUF_SYNC_WRITE; + ioctl(fd, DMA_BUF_IOCTL_SYNC, &sync); + } + return; + } + const unsigned char *p = map; + fprintf(stderr, "tva-export probe plane=%d bytes=%02x %02x %02x %02x %02x %02x %02x %02x\n", + plane, p[0], p[1], p[2], p[3], p[4], p[5], p[6], p[7]); + munmap(map, map_size); + if (sync_started) { + sync.flags = DMA_BUF_SYNC_END | DMA_BUF_SYNC_WRITE; + if (ioctl(fd, DMA_BUF_IOCTL_SYNC, &sync) < 0) + fprintf(stderr, "tva-export probe plane=%d sync end failed errno=%d\n", + plane, errno); + } +} +#endif + #ifdef _WIN32 #include "frontend/winsys_handle.h" #include @@ -103,20 +229,47 @@ vlVaRemoveDpbSurface(vlVaSurface *surf, VASurfaceID id) void vlVaDestroySurface(vlVaDriver *drv, vlVaSurface *surf) { - if (surf->buffer) - surf->buffer->destroy(surf->buffer); - if (surf->pipe_fence) - drv->pipe->screen->fence_reference(drv->pipe->screen, &surf->pipe_fence, NULL); - if (surf->ctx) { - assert(_mesa_set_search(surf->ctx->surfaces, surf)); - _mesa_set_remove_key(surf->ctx->surfaces, surf); - if (surf->fence && surf->ctx->decoder && surf->ctx->decoder->destroy_fence) { - surf->ctx->decoder->destroy_fence(surf->ctx->decoder, surf->fence); - surf->fence = NULL; + vlVaContext *context; + + if (!surf) + return; + + /* Imported PRIME surfaces may retain a producer association until the + * compositor destroys them. Clear reverse links before releasing the + * producer object so a later VPP submission cannot dereference it. */ + if (drv && drv->surfaces) { + set_foreach(drv->surfaces, entry) { + vlVaSurface *other = (vlVaSurface *)entry->key; + if (other && other != surf && other->sync_surface == surf) + other->sync_surface = NULL; } + _mesa_set_remove_key(drv->surfaces, surf); } - if (surf->fence && drv->proc && drv->proc->destroy_fence) + + context = surf->ctx; + if (context) + mtx_lock(&context->mutex); + + if (surf->fence && context && context->decoder && + context->decoder->destroy_fence) { + context->decoder->destroy_fence(context->decoder, surf->fence); + surf->fence = NULL; + } + if (surf->fence && drv->proc && drv->proc->destroy_fence) { drv->proc->destroy_fence(drv->proc, surf->fence); + surf->fence = NULL; + } + if (surf->pipe_fence) + drv->pipe->screen->fence_reference(drv->pipe->screen, + &surf->pipe_fence, NULL); + if (surf->buffer) + surf->buffer->destroy(surf->buffer); + + if (context) { + assert(_mesa_set_search(context->surfaces, surf)); + _mesa_set_remove_key(context->surfaces, surf); + mtx_unlock(&context->mutex); + } if (surf->coded_buf) surf->coded_buf->coded_surf = NULL; util_dynarray_fini(&surf->subpics); @@ -140,8 +293,11 @@ vlVaDestroySurfaces(VADriverContextP ctx, VASurfaceID *surface_list, int num_sur mtx_unlock(&drv->mutex); return VA_STATUS_ERROR_INVALID_SURFACE; } - if (surf->ctx && surf->is_dpb) + if (surf->ctx && surf->is_dpb) { + mtx_lock(&surf->ctx->mutex); vlVaRemoveDpbSurface(surf, surface_list[i]); + mtx_unlock(&surf->ctx->mutex); + } vlVaDestroySurface(drv, surf); handle_table_remove(drv->htab, surface_list[i]); } @@ -150,26 +306,25 @@ vlVaDestroySurfaces(VADriverContextP ctx, VASurfaceID *surface_list, int num_sur return VA_STATUS_SUCCESS; } -static VAStatus -_vlVaSyncSurface(VADriverContextP ctx, VASurfaceID render_target, uint64_t timeout_ns) +VAStatus +vlVaSyncSurfaceObjectLocked(vlVaDriver *drv, vlVaSurface *surf, + uint64_t timeout_ns) { - vlVaDriver *drv; vlVaContext *context; - vlVaSurface *surf; struct pipe_fence_handle *fence; - if (!ctx) - return VA_STATUS_ERROR_INVALID_CONTEXT; - - drv = VL_VA_DRIVER(ctx); - if (!drv) - return VA_STATUS_ERROR_INVALID_CONTEXT; - - mtx_lock(&drv->mutex); - surf = handle_table_get(drv->htab, render_target); - if (!surf) { - mtx_unlock(&drv->mutex); + if (!drv || !surf) return VA_STATUS_ERROR_INVALID_SURFACE; + + /* A PRIME-imported VPP source has no VA context of its own. If it was + * created from a decoder export, wait for that producer before examining + * the imported resource. */ + if (surf->sync_surface && surf->sync_surface != surf && + surf->prime_fence && surf->sync_surface->fence == surf->prime_fence) { + VAStatus sync_status = vlVaSyncSurfaceObjectLocked( + drv, surf->sync_surface, timeout_ns); + if (sync_status != VA_STATUS_SUCCESS) + return sync_status; } if (surf->coded_buf) { @@ -180,10 +335,17 @@ _vlVaSyncSurface(VADriverContextP ctx, VASurfaceID render_target, uint64_t timeo fence = surf->fence; } + if (vl_va_export_debug_enabled()) + fprintf(stderr, "tva-export sync inspect surf=%p ctx=%p fence=%p coded=%p pipe=%p\n", + (void *)surf, (void *)context, (void *)fence, + (void *)surf->coded_buf, (void *)surf->pipe_fence); + if (surf->pipe_fence) { struct pipe_screen *pscreen = drv->pipe->screen; + TVA_EXPORT_LOG("sync pipe fence=%p timeout=%" PRIu64 "\n", + (void *)surf->pipe_fence, timeout_ns); if (!pscreen->fence_finish(pscreen, NULL, surf->pipe_fence, timeout_ns)) { - mtx_unlock(&drv->mutex); + TVA_EXPORT_LOG("sync pipe fence timed out\n"); return VA_STATUS_ERROR_TIMEDOUT; } pscreen->fence_reference(pscreen, &surf->pipe_fence, NULL); @@ -191,22 +353,60 @@ _vlVaSyncSurface(VADriverContextP ctx, VASurfaceID render_target, uint64_t timeo /* No outstanding operation: nothing to do. */ if (!fence) { - mtx_unlock(&drv->mutex); + TVA_EXPORT_LOG("sync no decoder fence\n"); return VA_STATUS_SUCCESS; } - if (!context || !context->decoder) { - mtx_unlock(&drv->mutex); + if (!context || !context->decoder || !context->decoder->fence_wait) { + TVA_EXPORT_LOG("sync invalid context=%p decoder=%p\n", + (void *)context, + context ? (void *)context->decoder : NULL); return VA_STATUS_ERROR_INVALID_CONTEXT; } mtx_lock(&context->mutex); mtx_unlock(&drv->mutex); int ret = context->decoder->fence_wait(context->decoder, fence, timeout_ns); + TVA_EXPORT_LOG("sync decoder fence=%p result=%d\n", (void *)fence, ret); mtx_unlock(&context->mutex); + mtx_lock(&drv->mutex); return ret ? VA_STATUS_SUCCESS : VA_STATUS_ERROR_TIMEDOUT; } +static VAStatus +_vlVaSyncSurface(VADriverContextP ctx, VASurfaceID render_target, uint64_t timeout_ns) +{ + vlVaDriver *drv; + vlVaSurface *surf; + + if (!ctx) + return VA_STATUS_ERROR_INVALID_CONTEXT; + + drv = VL_VA_DRIVER(ctx); + if (!drv) + return VA_STATUS_ERROR_INVALID_CONTEXT; + + mtx_lock(&drv->mutex); + surf = handle_table_get(drv->htab, render_target); + if (!surf) { + mtx_unlock(&drv->mutex); + return VA_STATUS_ERROR_INVALID_SURFACE; + } + + /* Some Chromium paths call vaSyncSurface before they have submitted the + * next AV1 temporal unit. The bridge's reader thread cannot receive that + * first output until the pipeline is allowed to advance; defer the actual + * producer wait to VPP when this diagnostic override is enabled. */ + if (vlVaSurfaceNoWait()) { + mtx_unlock(&drv->mutex); + return VA_STATUS_SUCCESS; + } + + VAStatus ret = vlVaSyncSurfaceObjectLocked(drv, surf, timeout_ns); + mtx_unlock(&drv->mutex); + return ret; +} + VAStatus vlVaSyncSurface(VADriverContextP ctx, VASurfaceID render_target) { @@ -291,6 +491,9 @@ vlVaQuerySurfaceAttributes(VADriverContextP ctx, VAConfigID config_id, struct pipe_screen *pscreen; int i; + fprintf(stderr, "tva-va: query surface attrs config=%u list=%p count=%u\n", + config_id, (void *)attrib_list, num_attribs ? *num_attribs : 0); + if (config_id == VA_INVALID_ID) return VA_STATUS_ERROR_INVALID_CONFIG; @@ -299,6 +502,7 @@ vlVaQuerySurfaceAttributes(VADriverContextP ctx, VAConfigID config_id, if (!attrib_list) { *num_attribs = VL_VA_MAX_IMAGE_FORMATS + VASurfaceAttribCount; + fprintf(stderr, "tva-va: query surface attrs count=%u\n", *num_attribs); return VA_STATUS_SUCCESS; } @@ -477,6 +681,7 @@ vlVaQuerySurfaceAttributes(VADriverContextP ctx, VAConfigID config_id, if (i > *num_attribs) { *num_attribs = i; FREE(attribs); + fprintf(stderr, "tva-va: query surface attrs too small need=%d\n", i); return VA_STATUS_ERROR_MAX_NUM_EXCEEDED; } @@ -484,6 +689,8 @@ vlVaQuerySurfaceAttributes(VADriverContextP ctx, VAConfigID config_id, memcpy(attrib_list, attribs, i * sizeof(VASurfaceAttrib)); FREE(attribs); + fprintf(stderr, "tva-va: query surface attrs success count=%d\n", i); + return VA_STATUS_SUCCESS; } @@ -525,6 +732,8 @@ surface_from_external_memory(VADriverContextP ctx, vlVaSurface *surface, res_templ.depth0 = 1; res_templ.array_size = 1; res_templ.bind = PIPE_BIND_SAMPLER_VIEW; + if (!util_format_is_yuv(templat->buffer_format)) + res_templ.bind |= PIPE_BIND_RENDER_TARGET; res_templ.usage = PIPE_USAGE_DEFAULT; memset(&whandle, 0, sizeof(struct winsys_handle)); @@ -576,6 +785,91 @@ surface_from_external_memory(VADriverContextP ctx, vlVaSurface *surface, return result; } +vlVaSurface * +surface_find_prime_producer_for_surface(vlVaDriver *drv, + vlVaSurface *import_surface) +{ + struct pipe_screen *pscreen = drv && drv->vscreen ? + drv->vscreen->pscreen : NULL; + struct pipe_surface *import_surfaces; + + if (!drv || !pscreen || !drv->surfaces || !import_surface || + !import_surface->buffer) + return NULL; + + import_surfaces = import_surface->buffer->get_surfaces(import_surface->buffer); + if (!import_surfaces) + return NULL; + + /* PRoot does not reliably implement os_same_file_description for + * dma-bufs. Duplicated dma-buf descriptors still share device/inode. */ + for (unsigned import_plane = 0; import_plane < VL_MAX_SURFACES; + import_plane++) { + struct pipe_resource *import_resource = import_surfaces[import_plane].texture; + if (!import_resource) + continue; + + struct winsys_handle import_handle; + memset(&import_handle, 0, sizeof(import_handle)); + import_handle.type = WINSYS_HANDLE_TYPE_FD; + if (!pscreen->resource_get_handle(pscreen, drv->pipe, import_resource, + &import_handle, 0)) + continue; + + struct stat import_stat; + bool import_stat_valid = fstat(import_handle.handle, &import_stat) == 0; + + set_foreach(drv->surfaces, entry) { + vlVaSurface *candidate = (vlVaSurface *)entry->key; + if (!candidate || candidate == import_surface || + !candidate->buffer) + continue; + + struct pipe_surface *candidate_surfaces = + candidate->buffer->get_surfaces(candidate->buffer); + if (!candidate_surfaces) + continue; + + for (unsigned candidate_plane = 0; candidate_plane < VL_MAX_SURFACES; + candidate_plane++) { + struct pipe_resource *candidate_resource = + candidate_surfaces[candidate_plane].texture; + if (!candidate_resource) + continue; + + struct winsys_handle candidate_handle; + memset(&candidate_handle, 0, sizeof(candidate_handle)); + candidate_handle.type = WINSYS_HANDLE_TYPE_FD; + if (!pscreen->resource_get_handle(pscreen, drv->pipe, + candidate_resource, + &candidate_handle, 0)) + continue; + + struct stat candidate_stat; + bool match = import_stat_valid && + fstat(candidate_handle.handle, &candidate_stat) == 0 && + import_stat.st_dev == candidate_stat.st_dev && + import_stat.st_ino == candidate_stat.st_ino; + close(candidate_handle.handle); + if (match) { + close(import_handle.handle); + if (getenv("DMD_VA_PROBE")) + fprintf(stderr, "tva-va: prime producer matched import=%p " + "producer=%p plane=%u fence=%p ctx=%p\n", + (void *)import_surface, (void *)candidate, + candidate_plane, (void *)candidate->fence, + (void *)candidate->ctx); + return candidate; + } + } + } + + close(import_handle.handle); + } + + return NULL; +} + static VAStatus surface_from_prime(VADriverContextP ctx, vlVaSurface *surface, VADRMPRIMESurfaceDescriptor *desc, int mem_type, @@ -594,6 +888,30 @@ surface_from_prime(VADriverContextP ctx, vlVaSurface *surface, pscreen = VL_VA_PSCREEN(ctx); drv = VL_VA_DRIVER(ctx); + if (getenv("DMD_VA_PROBE")) { + fprintf(stderr, "tva-va: import prime surface=%p type=%#x desc=%p " + "fourcc=%#x size=%ux%u objects=%u layers=%u\n", + (void *)surface, mem_type, (void *)desc, + desc ? desc->fourcc : 0, desc ? desc->width : 0, + desc ? desc->height : 0, desc ? desc->num_objects : 0, + desc ? desc->num_layers : 0); + if (desc) { + for (unsigned i = 0; i < desc->num_objects; i++) + fprintf(stderr, "tva-va: import prime object[%u] fd=%d size=%u mod=%#llx\n", + i, desc->objects[i].fd, desc->objects[i].size, + (unsigned long long)desc->objects[i].drm_format_modifier); + for (unsigned i = 0; i < desc->num_layers; i++) { + fprintf(stderr, "tva-va: import prime layer[%u] fmt=%#x planes=%u\n", + i, desc->layers[i].drm_format, desc->layers[i].num_planes); + for (unsigned j = 0; j < desc->layers[i].num_planes; j++) + fprintf(stderr, "tva-va: import prime layer[%u].plane[%u] obj=%u " + "pitch=%u offset=%u\n", i, j, + desc->layers[i].object_index[j], + desc->layers[i].pitch[j], desc->layers[i].offset[j]); + } + } + } + if (!desc || desc->num_layers >= 4 ||desc->num_objects == 0) return VA_STATUS_ERROR_INVALID_PARAMETER; @@ -637,6 +955,8 @@ surface_from_prime(VADriverContextP ctx, vlVaSurface *surface, res_templ.depth0 = 1; res_templ.array_size = 1; res_templ.bind = PIPE_BIND_SAMPLER_VIEW; + if (!util_format_is_yuv(templat->buffer_format)) + res_templ.bind |= PIPE_BIND_RENDER_TARGET; res_templ.usage = PIPE_USAGE_DEFAULT; res_templ.format = templat->buffer_format; @@ -692,11 +1012,18 @@ surface_from_prime(VADriverContextP ctx, vlVaSurface *surface, } surface->buffer->contiguous_planes = true; + surface->is_prime_import = true; for (uint32_t i = 1; i < desc->num_objects; i++) { if (os_same_file_description(desc->objects[0].fd, desc->objects[i].fd) != 0) surface->buffer->contiguous_planes = false; } + surface->sync_surface = surface_find_prime_producer_for_surface(drv, surface); + surface->prime_fence = surface->sync_surface ? surface->sync_surface->fence : NULL; + if (getenv("DMD_VA_PROBE")) + fprintf(stderr, "tva-va: prime import producer=%p fence=%p\n", + (void *)surface->sync_surface, (void *)surface->prime_fence); + return VA_STATUS_SUCCESS; fail: @@ -754,6 +1081,13 @@ vlVaHandleSurfaceAllocate(vlVaDriver *drv, vlVaSurface *surface, struct pipe_surface *surfaces; unsigned i; + fprintf(stderr, "tva-va: allocate surface=%p format=%d size=%ux%u modifiers=%u\n", + (void *)surface, + surface ? surface->templat.buffer_format : -1, + surface ? surface->templat.width : 0, + surface ? surface->templat.height : 0, + modifiers_count); + if (modifiers_count > 0) { if (!drv->pipe->create_video_buffer_with_modifiers) return VA_STATUS_ERROR_ATTR_NOT_SUPPORTED; @@ -765,7 +1099,20 @@ vlVaHandleSurfaceAllocate(vlVaDriver *drv, vlVaSurface *surface, surface->buffer = drv->pipe->create_video_buffer(drv->pipe, &surface->templat); } if (!surface->buffer) + { + fprintf(stderr, "tva-va: surface allocation failed\n"); return VA_STATUS_ERROR_ALLOCATION_FAILED; + } + + fprintf(stderr, "tva-va: surface allocation ready buffer=%p\n", + (void *)surface->buffer); + + /* The termux-va bridge fills every plane before exporting a decoded + * surface. Avoid submitting the generic Gallium clear for these linear + * resources: the clear fence can otherwise serialize the independent + * KGSL Gallium and Vulkan queues before the first decoded frame. */ + if (vl_va_bridge_skip_clear()) + return VA_STATUS_SUCCESS; if (drv->pipe->screen->get_video_param(drv->pipe->screen, PIPE_VIDEO_PROFILE_UNKNOWN, @@ -805,6 +1152,7 @@ vlVaGetSurfaceBuffer(vlVaDriver *drv, vlVaSurface *surface) if (surface->buffer) return surface->buffer; vlVaHandleSurfaceAllocate(drv, surface, NULL, 0); + fprintf(stderr, "tva-va: get surface buffer=%p\n", (void *)surface->buffer); return surface->buffer; } @@ -896,6 +1244,9 @@ vlVaCreateSurfaces2(VADriverContextP ctx, unsigned int format, if (!ctx) return VA_STATUS_ERROR_INVALID_CONTEXT; + fprintf(stderr, "tva-va: create surfaces format=%#x size=%ux%u count=%u attrs=%u\n", + format, width, height, num_surfaces, num_attribs); + if (!(width && height)) return VA_STATUS_ERROR_INVALID_IMAGE_FORMAT; @@ -994,6 +1345,13 @@ vlVaCreateSurfaces2(VADriverContextP ctx, unsigned int format, } } + if (getenv("DMD_VA_PROBE")) { + fprintf(stderr, "tva-va: surface attrs resolved type=%#x expected=%#x " + "prime=%p ext=%p modifiers=%u bind=%#x\n", memory_type, + expected_fourcc, (void *)prime_desc, (void *)memory_attribute, + modifiers_count, templat.bind); + } + switch (memory_type) { case VA_SURFACE_ATTRIB_MEM_TYPE_VA: break; @@ -1107,6 +1465,7 @@ vlVaCreateSurfaces2(VADriverContextP ctx, unsigned int format, vaStatus = VA_STATUS_ERROR_ALLOCATION_FAILED; goto destroy_surf; } + _mesa_set_add(drv->surfaces, surf); } if (memory_type != VA_SURFACE_ATTRIB_MEM_TYPE_VA) @@ -1206,11 +1565,25 @@ vlVaExportSurfaceHandle(VADriverContextP ctx, drv = VL_VA_DRIVER(ctx); screen = VL_VA_PSCREEN(ctx); + + /* VA clients such as FFmpeg export surfaces before importing them into + * another API. Make the bridge's staged frame copy visible before a + * read-capable DMA-BUF export; WRITE_ONLY exports are destinations. */ + if (!(flags & VA_EXPORT_SURFACE_WRITE_ONLY) && !vl_va_export_no_wait()) { + ret = _vlVaSyncSurface(ctx, surface_id, VA_TIMEOUT_INFINITE); + if (ret != VA_STATUS_SUCCESS) + return ret; + } + mtx_lock(&drv->mutex); surf = handle_table_get(drv->htab, surface_id); + if (!surf) { + mtx_unlock(&drv->mutex); + return VA_STATUS_ERROR_INVALID_SURFACE; + } vlVaGetSurfaceBuffer(drv, surf); - if (!surf || !surf->buffer) { + if (!surf->buffer) { mtx_unlock(&drv->mutex); return VA_STATUS_ERROR_INVALID_SURFACE; } @@ -1222,6 +1595,12 @@ vlVaExportSurfaceHandle(VADriverContextP ctx, surfaces = surf->buffer->get_surfaces(surf->buffer); + if (vl_va_export_debug_enabled()) + fprintf(stderr, "tva-export surface=%#x format=%d size=%ux%u flags=%#x mem=%#x p0=%p p1=%p\n", + surface_id, surf->buffer->buffer_format, surf->templat.width, + surf->templat.height, flags, mem_type, + surfaces[0].texture, surfaces[1].texture); + usage = 0; if (flags & VA_EXPORT_SURFACE_WRITE_ONLY) usage |= PIPE_HANDLE_USAGE_FRAMEBUFFER_WRITE; @@ -1275,16 +1654,26 @@ vlVaExportSurfaceHandle(VADriverContextP ctx, if (!screen->resource_get_handle(screen, drv->pipe, resource, &whandle, usage)) { + if (vl_va_export_debug_enabled()) + fprintf(stderr, "tva-export resource_get_handle failed plane=%d res=%p\n", + p, (void *)resource); ret = VA_STATUS_ERROR_INVALID_SURFACE; goto fail; } + if (vl_va_export_debug_enabled()) + fprintf(stderr, "tva-export plane=%d fd=%d stride=%u offset=%u size=%" PRIu64 " mod=%#" PRIx64 " fmt=%#x\n", + p, whandle.handle, whandle.stride, whandle.offset, + whandle.size, whandle.modifier, drm_format); + vl_va_export_probe_fd(whandle.handle, whandle.size, p); + /* If this plane shares storage with previous one, we can reuse * the existing object (fd) instead of adding new one. */ bool same_object = desc->num_objects && - os_same_file_description(desc->objects[desc->num_objects - 1].fd, - whandle.handle) == 0; + (surf->buffer->contiguous_planes || + os_same_file_description(desc->objects[desc->num_objects - 1].fd, + whandle.handle) == 0); if (!same_object) { desc->objects[desc->num_objects].fd = (int) whandle.handle; @@ -1344,6 +1733,11 @@ vlVaExportSurfaceHandle(VADriverContextP ctx, return VA_STATUS_SUCCESS; fail: +#ifndef _WIN32 + if (vl_va_export_debug_enabled()) + fprintf(stderr, "tva-export failed surface=%#x status=%#x objects=%u\n", + surface_id, ret, desc->num_objects); +#endif #ifndef _WIN32 for (i = 0; i < desc->num_objects; i++) close(desc->objects[i].fd); diff --git a/src/gallium/frontends/va/tva_bridge.c b/src/gallium/frontends/va/tva_bridge.c new file mode 100644 index 000000000000..1c367849cd11 --- /dev/null +++ b/src/gallium/frontends/va/tva_bridge.c @@ -0,0 +1,3817 @@ +/* + * SPDX-License-Identifier: GPL-3.0-or-later + * + * tva_bridge.c - the termux-va bridge: forwards VA-API decode work from the + * Mesa VA frontend over a Unix socket to the termux-va daemon running in + * Termux, which decodes with Android MediaCodec in hardware and returns + * NV12 frames. + * + * Copyright (C) 2026 lfdevs + * + * This program is free software: you can redistribute it and/or modify + * it under the terms of the GNU General Public License as published by + * the Free Software Foundation, version 3 of the License. + * + * ****************************************************************************** + * MODIFICATION NOTICE (GPL-3.0 section 5) + * + * Parts of this file are a MODIFIED version of vaapi-driver/src/decode.c, + * vaapi-driver/src/profiles.c, and vaapi-driver/src/hevc_bitstream.c from + * the droidspaces-media-decode project (Apache License, Version 2.0): the + * codec capability table, the pending pipeline-depth model, the + * is_param_set() unit classification, and the HEVC parameter-set writer were + * ported from there and relicensed under GPL-3.0. The Mesa-side wrappers and + * the fence/pending machinery are new code written for termux-va. + * ****************************************************************************** + * + * Architecture (Mesa 26.x VA frontend, new video API): + * + * vaRenderPicture -> frontend parses VA buffers, prepends H.264/HEVC + * start codes to slice data, then calls + * decode_bitstream(...) -> we ACCUMULATE the bytes + * vaEndPicture -> end_frame(...) -> we send the accumulated Annex B + * access unit as one daemon input unit and register a + * fence for the picture + * vaSyncSurface -> fence_wait(...) -> we pump frames from the daemon, + * stage the frame matching the picture's unit index, + * and copy the visible (cropped) region into the + * surface's plane resources on the caller's thread + * + * Threading model: the reader thread owns socket receives and stages complete + * frames under pend_mutex. end_frame runs on the application thread and + * publishes a pending record before sending output-producing units. The + * fence_wait callback remains on the application thread and copies staged + * planes through pipe_context after the reader signals the record. + * + * Unit-index pairing: the daemon tags every VCL input unit with an index + * (1-based, parameter sets excluded) and carries it back on the matching + * output frame. A picture's complete access unit therefore maps to one + * index; frames with unknown indices fall back to FIFO matching. + */ +#include "tva_bridge.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#ifndef _WIN32 +#include +#include +#include +#include +#include +#include +#include "drm-uapi/dma-buf.h" +#include "drm-uapi/drm_fourcc.h" +#ifdef __linux__ +#include +#endif +#include "frontend/drm_driver.h" +#endif + +#include "pipe/p_context.h" +#include "pipe/p_screen.h" +#include "pipe/p_video_codec.h" +#include "pipe/p_video_enums.h" +#include "pipe/p_state.h" + +#include "pipe-loader/pipe_loader.h" + +#include "c11/threads.h" +#include "util/cnd_monotonic.h" +#include "util/format/u_format.h" +#include "util/os_misc.h" +#include "util/os_time.h" +#include "util/timespec.h" +#include "util/u_debug.h" +#include "util/u_memory.h" +#include "util/u_video.h" +#include "vl/vl_video_buffer.h" +#include "vl/vl_compositor_proc.h" +#include "vl/vl_winsys.h" + +#include "tva_client.h" +#include "tva_protocol.h" +#include "av1_bitstream.h" + +/* Normal bridge pipeline depth. In SHM mode it MUST stay <= SHM_SLOTS (8, + * tva_protocol.h) or the daemon's slot pool stalls; the pending cache below + * may temporarily grow beyond this threshold while a decoder reorders output, + * but each SHM slot is released immediately after staging. */ +#define DMD_PIPELINE_DEPTH_MAX 32 +static unsigned tva_pipeline_depth_default = 6; + +/* ----------------------------------------------------------- activation */ +bool tva_bridge_active(void) +{ + const char *force = os_get_option("TERMUX_VA_BRIDGE"); + if (force && *force) { + return !(strcmp(force, "0") == 0 || strcmp(force, "false") == 0 || + strcmp(force, "off") == 0); + } + + const char *ep = os_get_option("TERMUX_VA_SOCKET"); + if (ep && *ep) + return true; + ep = os_get_option("TERMUX_VA_SOCKET_DIR"); + if (ep && *ep) + return true; + + /* Auto-detect: the daemon's default endpoint seen through the shared tmp */ + char buf[300]; + const char *def = tva_default_endpoint(buf, sizeof(buf)); + struct stat st; + return def && stat(def, &st) == 0 && S_ISSOCK(st.st_mode); +} + +static bool +tva_dbg(void) +{ + const char *e = getenv("DMD_VA_LOG"); + return e && e[0] == '1'; +} + +static int tva_dbg_seq; +#define TVA_TRACE(fmt, ...) \ + do { if (tva_dbg()) \ + fprintf(stderr, "tva#%d %s: " fmt "\n", tva_dbg_seq, __func__, ##__VA_ARGS__); \ + } while (0) + +/* ------------------------------------------------- screen video hooks */ +/* + * Drivers without a video path (freedreno, llvmpipe) leave + * get_video_param / is_video_format_supported NULL, which fails the VA + * frontend's init check. When the bridge is active it owns video decode, + * so its capability hooks replace generic 3D-driver hooks that would reject + * video formats before the bridge receives the request. */ + +static bool +tva_option_enabled(const char *name) +{ + const char *value = os_get_option(name); + + return value && *value && + (strcmp(value, "1") == 0 || strcmp(value, "true") == 0 || + strcmp(value, "on") == 0); +} + +static bool +tva_avc_disabled(void) +{ + return tva_option_enabled("TERMUX_VA_DISABLE_AVC") || + tva_option_enabled("TERMUX_VA_DISABLE_H264"); +} + +static bool +tva_hevc_disabled(void) +{ + return tva_option_enabled("TERMUX_VA_DISABLE_HEVC"); +} + +static bool +tva_vp9_disabled(void) +{ + return tva_option_enabled("TERMUX_VA_DISABLE_VP9"); +} + +static bool +tva_av1_disabled(void) +{ + return tva_option_enabled("TERMUX_VA_DISABLE_AV1"); +} + +static bool +tva_profile_supported(enum pipe_video_profile profile) +{ + switch (profile) { + case PIPE_VIDEO_PROFILE_MPEG4_AVC_CONSTRAINED_BASELINE: + case PIPE_VIDEO_PROFILE_MPEG4_AVC_MAIN: + case PIPE_VIDEO_PROFILE_MPEG4_AVC_HIGH: + return !tva_avc_disabled(); + case PIPE_VIDEO_PROFILE_HEVC_MAIN: + return !tva_hevc_disabled(); + case PIPE_VIDEO_PROFILE_VP9_PROFILE0: + return !tva_vp9_disabled(); + case PIPE_VIDEO_PROFILE_AV1_MAIN: + return !tva_av1_disabled(); + default: + return false; + } +} + +static int +tva_codec_id(enum pipe_video_profile profile) +{ + switch (u_reduce_video_profile(profile)) { + case PIPE_VIDEO_FORMAT_MPEG4_AVC: + return CODEC_H264; + case PIPE_VIDEO_FORMAT_HEVC: + return CODEC_HEVC; + case PIPE_VIDEO_FORMAT_VP9: + return CODEC_VP9; + case PIPE_VIDEO_FORMAT_AV1: + return CODEC_AV1; + default: + return -1; + } +} + +static int +tva_screen_get_video_param(struct pipe_screen *screen, + enum pipe_video_profile profile, + enum pipe_video_entrypoint entrypoint, + enum pipe_video_cap param) +{ + if (entrypoint == PIPE_VIDEO_ENTRYPOINT_PROCESSING) { + switch (param) { + case PIPE_VIDEO_CAP_SUPPORTED: + case PIPE_VIDEO_CAP_SUPPORTS_PROGRESSIVE: + return 1; + case PIPE_VIDEO_CAP_MIN_WIDTH: + case PIPE_VIDEO_CAP_MIN_HEIGHT: + case PIPE_VIDEO_CAP_VPP_MIN_INPUT_WIDTH: + case PIPE_VIDEO_CAP_VPP_MIN_INPUT_HEIGHT: + case PIPE_VIDEO_CAP_VPP_MIN_OUTPUT_WIDTH: + case PIPE_VIDEO_CAP_VPP_MIN_OUTPUT_HEIGHT: + return 1; + case PIPE_VIDEO_CAP_MAX_WIDTH: + case PIPE_VIDEO_CAP_VPP_MAX_INPUT_WIDTH: + case PIPE_VIDEO_CAP_VPP_MAX_OUTPUT_WIDTH: + return 8192; + case PIPE_VIDEO_CAP_MAX_HEIGHT: + case PIPE_VIDEO_CAP_VPP_MAX_INPUT_HEIGHT: + case PIPE_VIDEO_CAP_VPP_MAX_OUTPUT_HEIGHT: + return 4320; + case PIPE_VIDEO_CAP_VPP_ORIENTATION_MODES: + return PIPE_VIDEO_VPP_ORIENTATION_DEFAULT; + case PIPE_VIDEO_CAP_VPP_BLEND_MODES: + return PIPE_VIDEO_VPP_BLEND_MODE_NONE; + case PIPE_VIDEO_CAP_SUPPORTS_CONTIGUOUS_PLANES_MAP: + return 0; + default: + return 0; + } + } + + if ((entrypoint == PIPE_VIDEO_ENTRYPOINT_BITSTREAM || + entrypoint == PIPE_VIDEO_ENTRYPOINT_UNKNOWN) && + param == PIPE_VIDEO_CAP_SUPPORTS_PROGRESSIVE && + (profile == PIPE_VIDEO_PROFILE_UNKNOWN || tva_profile_supported(profile))) + return 1; + + if (entrypoint == PIPE_VIDEO_ENTRYPOINT_BITSTREAM && + tva_profile_supported(profile)) { + switch (param) { + case PIPE_VIDEO_CAP_SUPPORTED: + return 1; + case PIPE_VIDEO_CAP_MIN_WIDTH: + case PIPE_VIDEO_CAP_MIN_HEIGHT: + return 96; + case PIPE_VIDEO_CAP_MAX_WIDTH: + return 8192; + case PIPE_VIDEO_CAP_MAX_HEIGHT: + return 4320; + case PIPE_VIDEO_CAP_SUPPORTS_PROGRESSIVE: + return 1; + case PIPE_VIDEO_CAP_SKIP_CLEAR_SURFACE: + /* surfaces start uninitialized; every decode target is fully + * written by the first frame copy */ + return 1; + case PIPE_VIDEO_CAP_SUPPORTS_CONTIGUOUS_PLANES_MAP: + /* plane resources are separate textures; vaDeriveImage is not + * available (CPU consumers fall back to vaGetImage) */ + return 0; + default: + return 0; + } + } + + return 0; +} + +static bool +tva_screen_is_video_format_supported(struct pipe_screen *screen, + enum pipe_format format, + enum pipe_video_profile profile, + enum pipe_video_entrypoint entrypoint) +{ + if (entrypoint == PIPE_VIDEO_ENTRYPOINT_PROCESSING) { + /* Chromium's Linux VAAPI path uses VideoProc as a fallback when the + * decoded NV12 surface cannot be rendered directly. The bridge has + * no hardware VPP, but the Gallium compositor can convert the linear + * NV12 resources to the packed RGB surfaces used by the X11 ANGLE + * path. */ + switch (format) { + case PIPE_FORMAT_NV12: + case PIPE_FORMAT_R8G8B8A8_UNORM: + case PIPE_FORMAT_B8G8R8A8_UNORM: + case PIPE_FORMAT_R8G8B8X8_UNORM: + case PIPE_FORMAT_B8G8R8X8_UNORM: + case PIPE_FORMAT_A8R8G8B8_UNORM: + break; + default: + return false; + } + return vl_video_buffer_is_format_supported(screen, format, + PIPE_VIDEO_PROFILE_UNKNOWN, + entrypoint); + } + + if (entrypoint != PIPE_VIDEO_ENTRYPOINT_BITSTREAM || + !tva_profile_supported(profile)) + return false; + if (format != PIPE_FORMAT_NV12) + return false; + return vl_video_buffer_is_format_supported(screen, format, profile, + entrypoint); +} + +void +tva_bridge_screen_set_video_hooks(struct pipe_screen *screen) +{ + if (!screen) + return; + screen->get_video_param = tva_screen_get_video_param; + screen->is_video_format_supported = tva_screen_is_video_format_supported; +} + +/* ------------------------------------------------------- bridge codec */ +struct tva_fence; +struct tva_av1_frame; + +struct tva_pending { + bool in_use; + uint32_t unit_seq; /* access-unit index of the picture */ + bool ready; /* staged frame available */ + bool failed; /* session error: fence must not hang */ + bool copied; /* staging already written into the target */ + bool resource_notified; /* external write notification delivered */ + bool drop_on_fence_destroy; /* target was reused before this output */ + unsigned waiters; /* fence_wait callers holding this entry */ + struct pipe_resource *resources[2]; /* owned until the entry is reaped */ + uint8_t *staging; + size_t staging_size; + uint32_t frame_width; + uint32_t frame_height; + int stride; + int slice_height; + int crop_left; + int crop_top; + int crop_right; + int crop_bottom; + struct tva_fence *fence; +}; + +struct tva_fence { + struct tva_codec *codec; + struct tva_pending *slot; + bool failed; /* the associated pending entry was abandoned */ +}; + +/* AV1's VA-API descriptor does not carry refresh_frame_flags. Keep a whole + * temporal unit in hand until the next descriptor arrives: its reference map + * then reveals which slots contain the just-finished target, allowing us to + * reconstruct the exact refresh mask before submitting the unit. The legacy + * path submits hidden frames and their displayed frame together; the default + * hidden-output path intentionally splits them so MediaCodec returns every + * frame needed by show_existing_frame playback. */ +#define TVA_AV1_GROUP_MAX_FRAMES 64 +/* Keep surface reuse bounded even when MediaCodec is temporarily stalled. */ +#define TVA_FENCE_DESTROY_WAIT_MS 200 + +struct tva_av1_picture { + struct pipe_video_buffer *target; + struct tva_pending *pending; + VADecPictureParameterBufferAV1 picture; + struct dmd_av1_tile tiles[256]; + uint8_t *tile_data; + size_t tile_bytes; + unsigned tile_count; + uint8_t refresh_frame_flags; +}; + +struct tva_av1_frame { + bool valid; + struct tva_av1_picture pictures[TVA_AV1_GROUP_MAX_FRAMES]; + unsigned picture_count; + bool include_sequence; + bool have_show_frame; + bool hidden_since_show_frame; +}; + +struct tva_codec { + struct pipe_video_codec base; + + struct pipe_context *pipe; /* real context, for resource writes */ + + struct tva_session *sess; + + /* Bitstream accumulation for the picture being assembled + * (decode_bitstream copies VA buffers here; the frontend unmaps them + * after RenderPicture returns). */ + uint8_t *acc; + size_t acc_len; + size_t acc_cap; + + /* Pending ring, FIFO order */ + struct tva_pending pend[DMD_PIPELINE_DEPTH_MAX]; + unsigned pipeline_depth; + bool strict_pending; /* optional hard in-flight limit */ + unsigned pend_head; /* oldest entry */ + unsigned pend_count; + + uint64_t next_unit; /* index to assign to the next VCL unit (1-based) */ + uint64_t frames_done; /* staged frames counter (diagnostics) */ + + /* synthesized CSD cache: re-sent to the daemon only when it changes */ + uint8_t *csd; + size_t csd_len; + bool h264_pps_defaults_valid; + unsigned h264_pps_l0_default; + unsigned h264_pps_l1_default; + bool av1_sequence_sent; + struct tva_av1_frame av1_pending; + + bool broken; /* session error, further decodes fail */ + + /* reader thread machinery */ + mtx_t pend_mutex; + struct u_cnd_monotonic pend_cond; + bool quitting; + thrd_t reader; + bool reader_started; +}; + +static struct tva_codec * +tva_codec(struct pipe_video_codec *codec) +{ + return (struct tva_codec *)codec; +} + +/* ---------------------------- NALU classification (upstream port) */ +/* + * H.264 NAL unit type. Returns -1 when undeterminable. Ported from + * upstream src/decode-daemon.c: the start code (3 or 4 bytes) must be + * skipped before reading the header byte. + */ +static int +tva_nalu_type(const uint8_t *b, size_t len) +{ + size_t off = 0; + if (len >= 4 && b[0] == 0 && b[1] == 0 && b[2] == 0 && b[3] == 1) off = 4; + else if (len >= 3 && b[0] == 0 && b[1] == 0 && b[2] == 1) off = 3; + if (off == 0 || off >= len) return -1; + return b[off] & 0x1f; +} + +/* + * Whether the unit is a parameter set. The daemon does NOT count parameter + * sets in the unit index (vcl_in only advances for VCL units), so the + * bridge must apply the same classification to assign indices correctly. + * Ported from upstream src/decode-daemon.c is_param_set(). + */ +static bool +tva_is_param_set(int codec_id, const uint8_t *b, size_t len) +{ + if (codec_id == CODEC_H264) { + int t = tva_nalu_type(b, len); + return (t == 7 || t == 8); + } + if (codec_id == CODEC_HEVC) { + size_t off = 0; + if (len >= 4 && b[0] == 0 && b[1] == 0 && b[2] == 0 && b[3] == 1) off = 4; + else if (len >= 3 && b[0] == 0 && b[1] == 0 && b[2] == 1) off = 3; + if (off == 0 || off >= len) return false; + int t = (b[off] >> 1) & 0x3f; + return (t == 32 || t == 33 || t == 34); + } + return false; +} + +/* Locate the next 3-byte start code at or after `from` (the tail of a + * 4-byte code also matches); returns len when none. */ +static size_t +tva_next_start_code(const uint8_t *d, size_t len, size_t from) +{ + for (size_t i = from; i + 2 < len; i++) + if (d[i] == 0 && d[i + 1] == 0 && d[i + 2] == 1) + return i; + return len; +} + +/* ---------------------------- pending ring */ +static struct tva_pending * +tva_pend_oldest(struct tva_codec *c) +{ + if (!c->pend_count) + return NULL; + return &c->pend[c->pend_head]; +} + +/* Caller must hold pend_mutex. */ +static void +tva_pend_pop_locked(struct tva_codec *c) +{ + struct tva_pending *p = &c->pend[c->pend_head]; + if (p->fence) { + p->fence->failed = true; + p->fence->slot = NULL; + } + for (unsigned i = 0; i < ARRAY_SIZE(p->resources); i++) + pipe_resource_reference(&p->resources[i], NULL); + free(p->staging); + memset(p, 0, sizeof(*p)); + c->pend_head = (c->pend_head + 1) % DMD_PIPELINE_DEPTH_MAX; + c->pend_count--; +} + +/* Caller must hold pend_mutex. */ +static struct tva_pending * +tva_pend_find_locked(struct tva_codec *c, uint32_t unit_seq) +{ + for (unsigned i = 0; i < c->pend_count; i++) { + struct tva_pending *p = + &c->pend[(c->pend_head + i) % DMD_PIPELINE_DEPTH_MAX]; + if (p->in_use && !p->ready && unit_seq && p->unit_seq == unit_seq) + return p; + } + + /* A known sequence that is no longer pending was abandoned. Do not + * attach its late output to another surface. Legacy peers without PTS + * use FIFO matching because there is no other identity to compare. */ + return unit_seq ? NULL : tva_pend_oldest(c); +} + +/* Caller must hold pend_mutex. */ +static void +tva_fail_pending_locked(struct tva_codec *c) +{ + for (unsigned i = 0; i < c->pend_count; i++) { + struct tva_pending *p = + &c->pend[(c->pend_head + i) % DMD_PIPELINE_DEPTH_MAX]; + if (!p->ready) { + p->ready = true; + p->failed = true; + } + } +} + +static bool +tva_codec_is_broken(struct tva_codec *c) +{ + bool broken; + mtx_lock(&c->pend_mutex); + broken = c->broken; + mtx_unlock(&c->pend_mutex); + return broken; +} + +/* Caller must hold pend_mutex. */ +static void +tva_mark_broken_locked(struct tva_codec *c) +{ + c->broken = true; + tva_fail_pending_locked(c); + u_cnd_monotonic_broadcast(&c->pend_cond); +} + +static void +tva_mark_broken(struct tva_codec *c) +{ + mtx_lock(&c->pend_mutex); + tva_mark_broken_locked(c); + mtx_unlock(&c->pend_mutex); +} + +/* Caller must hold pend_mutex. */ +static void +tva_detach_fence_locked(struct tva_fence *fence, bool fail) +{ + if (!fence || !fence->slot) + return; + struct tva_pending *p = fence->slot; + if (p->fence == fence) + p->fence = NULL; + fence->slot = NULL; + if (fail) { + fence->failed = true; + if (!p->ready) { + p->ready = true; + p->failed = true; + } + } +} + +static bool +tva_copy_frame(struct tva_codec *c, struct tva_pending *p); + +/* Notify the driver after a frame has been written into the retained + * resources. The reader thread cannot call pipe_screen callbacks, so direct + * copies are notified by fence_wait on the application thread. */ +static void +tva_notify_frame_resources(struct tva_codec *c, struct tva_pending *p) +{ + if (!c || !p || p->resource_notified || !p->copied || !c->pipe || + !c->pipe->screen->resource_changed) + return; + + c->pipe->screen->resource_changed(c->pipe->screen, p->resources[0]); + c->pipe->screen->resource_changed(c->pipe->screen, p->resources[1]); + p->resource_notified = true; +} + +/* Caller must hold pend_mutex. A pending entry is reclaimable only after its + * frame has been staged and no fence waiter is still using it. The copy is + * deliberately performed here, on the application thread; the reader thread + * is restricted to socket I/O and staging. */ +static int +tva_pend_retire_oldest_locked(struct tva_codec *c) +{ + struct tva_pending *p = tva_pend_oldest(c); + if (!p || p->waiters || !p->ready) + return 0; + + if (!p->failed && !p->copied && p->staging) { + p->copied = tva_copy_frame(c, p); + TVA_TRACE("retire copy unit=%u result=%d", p->unit_seq, p->copied); + if (!p->copied) + p->failed = true; + } + tva_notify_frame_resources(c, p); + if (!p->copied && !p->failed) + return -1; + + if (p->fence) { + p->fence->failed = p->failed; + p->fence->slot = NULL; + p->fence = NULL; + } + tva_pend_pop_locked(c); + u_cnd_monotonic_broadcast(&c->pend_cond); + return 1; +} + +/* Caller must hold pend_mutex. */ +static struct tva_pending * +tva_pend_reserve_locked(struct tva_codec *c, uint32_t unit_seq, + struct pipe_video_buffer *target, + struct tva_fence *fence) +{ + while (c->pend_count >= c->pipeline_depth) { + int retired = tva_pend_retire_oldest_locked(c); + if (retired > 0) + continue; + if (retired < 0) { + tva_mark_broken_locked(c); + return NULL; + } + + /* The normal threshold is a scheduling hint, not a reason to drop a + * frame. If the decoder is still reordering the oldest output, let + * the host-side pending cache grow up to its fixed bound. */ + if (c->pend_count < DMD_PIPELINE_DEPTH_MAX && !c->strict_pending) + break; + + struct tva_pending *oldest = tva_pend_oldest(c); + if (!oldest) { + tva_mark_broken_locked(c); + return NULL; + } + + /* The cache is full. Wait for the oldest frame to become available, + * then retire it on this application thread before reserving space. */ + uint64_t deadline = os_time_get_nano() + + (uint64_t)SHM_SLOT_WAIT_MS * 1000000ull; + while (!oldest->ready || oldest->waiters) { + if (c->broken) { + tva_fail_pending_locked(c); + break; + } + uint64_t now = os_time_get_nano(); + if (now >= deadline) { + tva_mark_broken_locked(c); + return NULL; + } + struct timespec ts; + timespec_from_nsec(&ts, MIN2(deadline, now + 200000000ull)); + u_cnd_monotonic_timedwait(&c->pend_cond, &c->pend_mutex, &ts); + oldest = tva_pend_oldest(c); + if (!oldest) + break; + } + + retired = tva_pend_retire_oldest_locked(c); + if (retired <= 0) { + tva_mark_broken_locked(c); + return NULL; + } + } + + struct tva_pending *p = + &c->pend[(c->pend_head + c->pend_count) % DMD_PIPELINE_DEPTH_MAX]; + memset(p, 0, sizeof(*p)); + p->in_use = true; + p->unit_seq = unit_seq; + if (target) { + struct pipe_resource *res[4] = {0}; + target->get_resources(target, res); + for (unsigned i = 0; i < ARRAY_SIZE(p->resources); i++) { + if (res[i]) + pipe_resource_reference(&p->resources[i], res[i]); + } + } + p->fence = fence; + if (fence) { + fence->codec = c; + fence->slot = p; + } + c->pend_count++; + return p; +} + +/* AV1 show_existing_frame packets do not enter the VA decode callbacks. A + * hidden reference picture therefore has no consumer-visible callback even + * though its surface must still contain decoded pixels before a later + * show_existing_frame can display it. The Qualcomm decoder only returns + * output buffers for show_frame pictures, so mark hidden pictures as shown + * in the reconstructed stream and pair those extra output buffers with their + * original VA surfaces. DMD_AV1_OUTPUT_HIDDEN=0 restores the legacy path. */ +static bool +tva_av1_output_hidden(void) +{ + const char *e = getenv("DMD_AV1_OUTPUT_HIDDEN"); + if (!e || !*e) + return true; + return !(!strcmp(e, "0") || !strcmp(e, "false") || + !strcmp(e, "off")); +} + +static bool +tva_av1_synthetic_show_existing(void) +{ + const char *e = getenv("DMD_AV1_SYNTHETIC_SHOW"); + return e && (!strcmp(e, "1") || !strcmp(e, "true") || + !strcmp(e, "on")); +} + +static bool +tva_av1_inline_show_existing(void) +{ + const char *e = getenv("DMD_AV1_INLINE_SHOW"); + return e && (!strcmp(e, "1") || !strcmp(e, "true") || + !strcmp(e, "on")); +} + +static bool +tva_av1_strict_pending(void) +{ + const char *e = getenv("DMD_AV1_STRICT_PENDING"); + return e && (!strcmp(e, "1") || !strcmp(e, "true") || + !strcmp(e, "on")); +} + +static unsigned +tva_av1_fence_destroy_wait_ms(void) +{ + const char *e = getenv("DMD_AV1_FENCE_WAIT_MS"); + if (!e || !*e) + return TVA_FENCE_DESTROY_WAIT_MS; + + char *end = NULL; + unsigned long value = strtoul(e, &end, 10); + if (end == e || *end || value > 5000) + return TVA_FENCE_DESTROY_WAIT_MS; + return (unsigned)value; +} + +#if defined(__linux__) +static bool tva_drm_render_node_present(void); +#endif + +static bool +tva_cpu_copy_enabled(void) +{ + const char *e = getenv("DMD_VA_CPU_COPY"); + if (e && *e) + return !(strcmp(e, "0") == 0 || strcmp(e, "false") == 0 || + strcmp(e, "off") == 0); + + /* GPU uploads through imported KGSL dma-bufs are not reliably visible to + * the consumer in a PRoot container, which has no DRM render node. Use a + * direct CPU copy there so the dma-buf exporter can complete the handoff; + * DMD_VA_CPU_COPY=0 remains available for explicit GPU-upload diagnostics. */ +#if defined(__linux__) + const char *backend = getenv("TERMUX_VA_GPU_BACKEND"); + if ((!backend || !*backend || !strcmp(backend, "auto") || + !strcmp(backend, "kgsl")) && access("/dev/kgsl-3d0", R_OK) == 0 && + !tva_drm_render_node_present()) + return true; +#endif + return true; +} + +#if defined(__linux__) +static bool +tva_drm_render_node_present(void) +{ + DIR *dir = opendir("/dev/dri"); + if (!dir) + return false; + + bool present = false; + struct dirent *entry; + while ((entry = readdir(dir)) != NULL) { + if (!strncmp(entry->d_name, "renderD", 7)) { + present = true; + break; + } + } + closedir(dir); + return present; +} +#endif + +/* Chromium's Vulkan importer currently expects all NV12 planes to refer to + * one dma-buf object. The normal Gallium video-buffer allocator creates one + * object per plane, which is valid for VA-API but cannot be consumed by that + * importer in a PRoot environment. It can be overridden explicitly, and is + * enabled automatically for KGSL-only containers without a DRM render node. */ +static bool +tva_contiguous_dmabuf_enabled(void) +{ + const char *e = getenv("DMD_VA_CONTIGUOUS_DMABUF"); + if (!e || !*e) + e = getenv("TERMUX_VA_CONTIGUOUS_DMABUF"); + if (e && *e) + return !strcmp(e, "1") || !strcmp(e, "true") || !strcmp(e, "on"); + +#if defined(__linux__) + /* PRoot exposes KGSL directly but has no DRM render node. Chromium's + * native-pixmap importer accepts only one NV12 dma-buf in that setup, so + * select the shared-object layout automatically when no override is set. */ + const char *backend = getenv("TERMUX_VA_GPU_BACKEND"); + if ((!backend || !*backend || !strcmp(backend, "auto") || + !strcmp(backend, "kgsl")) && + access("/dev/kgsl-3d0", R_OK) == 0 && + !tva_drm_render_node_present()) + return true; +#endif + + return false; +} + +#ifndef _WIN32 +#if defined(__linux__) +static int +tva_alloc_dmabuf(size_t size) +{ + if (!size) + return -1; + + int heap_fd = open("/dev/dma_heap/system", O_RDONLY | O_CLOEXEC); + if (heap_fd < 0) + return -1; + + struct dma_heap_allocation_data alloc = { + .len = size, + .fd_flags = O_RDWR | O_CLOEXEC, + }; + int ret = ioctl(heap_fd, DMA_HEAP_IOCTL_ALLOC, &alloc); + int saved_errno = errno; + close(heap_fd); + if (ret < 0) { + errno = saved_errno; + return -1; + } + + return (int)alloc.fd; +} + +/* Allocate a linear NV12 buffer backed by one dma-buf and import each plane + * as a separate Gallium resource with an explicit offset. The resources + * intentionally have independent KGSL BO wrappers, but their exported FDs + * still refer to the same dma-buf file description. */ +static struct pipe_video_buffer * +tva_create_contiguous_video_buffer(struct pipe_context *context, + const struct pipe_video_buffer *templat) +{ + struct pipe_screen *screen = context ? context->screen : NULL; + struct pipe_video_buffer bridge_templ; + struct pipe_resource *resources[VL_NUM_COMPONENTS] = {0}; + enum pipe_format formats[VL_NUM_COMPONENTS] = {0}; + struct pipe_resource res_templ; + enum pipe_video_chroma_format chroma; + const unsigned page_size = 4096; + unsigned y_stride, y_height, uv_height; + size_t y_size, uv_size, uv_offset, total_size; + int dmabuf = -1; + struct pipe_video_buffer *result = NULL; + + if (!screen || !screen->resource_from_handle || + !templat || templat->buffer_format != PIPE_FORMAT_NV12 || + templat->interlaced) + return NULL; + + bridge_templ = *templat; + bridge_templ.bind |= PIPE_BIND_SHARED | PIPE_BIND_LINEAR; + bridge_templ.width = align(templat->width, VL_MACROBLOCK_WIDTH); + bridge_templ.height = align(templat->height, VL_MACROBLOCK_HEIGHT); + chroma = pipe_format_to_chroma_format(bridge_templ.buffer_format); + + vl_get_video_buffer_formats(screen, bridge_templ.buffer_format, formats); + if (formats[0] == PIPE_FORMAT_NONE || formats[1] == PIPE_FORMAT_NONE) + return NULL; + + memset(&res_templ, 0, sizeof(res_templ)); + vl_video_buffer_template(&res_templ, &bridge_templ, formats[0], 1, 1, + PIPE_USAGE_DEFAULT, 0, chroma); + y_stride = align(res_templ.width0 * util_format_get_blocksize(res_templ.format), + 64); + y_height = res_templ.height0; + + vl_video_buffer_template(&res_templ, &bridge_templ, formats[1], 1, 1, + PIPE_USAGE_DEFAULT, 1, chroma); + uv_height = res_templ.height0; + + if (!y_stride || !y_height || !uv_height || + y_height > SIZE_MAX / y_stride || + uv_height > SIZE_MAX / y_stride) + return NULL; + + y_size = (size_t)y_stride * y_height; + uv_size = (size_t)y_stride * uv_height; + uv_offset = align(y_size, page_size); + if (uv_offset < y_size || uv_size > SIZE_MAX - uv_offset) + return NULL; + total_size = align(uv_offset + uv_size, page_size); + if (total_size < uv_offset + uv_size || total_size > UINT32_MAX) + return NULL; + + dmabuf = tva_alloc_dmabuf(total_size); + if (dmabuf < 0) + return NULL; + + for (unsigned plane = 0; plane < 2; plane++) { + const enum pipe_format format = formats[plane]; + const size_t offset = plane ? uv_offset : 0; + int import_fd; + + vl_video_buffer_template(&res_templ, &bridge_templ, format, 1, 1, + PIPE_USAGE_DEFAULT, plane, chroma); + res_templ.bind |= PIPE_BIND_SHARED | PIPE_BIND_LINEAR; + + struct winsys_handle whandle = { + .type = WINSYS_HANDLE_TYPE_FD, + .plane = plane, + .size = (uint64_t)total_size, + .stride = y_stride, + .offset = offset, + .format = format, + .modifier = DRM_FORMAT_MOD_LINEAR, + }; + import_fd = dup(dmabuf); + if (import_fd < 0) + goto fail; + whandle.handle = import_fd; + resources[plane] = screen->resource_from_handle( + screen, &res_templ, &whandle, + PIPE_HANDLE_USAGE_FRAMEBUFFER_WRITE); + close(import_fd); + if (!resources[plane]) + goto fail; + } + + result = vl_video_buffer_create_ex2(context, &bridge_templ, resources); + if (!result) + goto fail; + result->contiguous_planes = true; + close(dmabuf); + if (getenv("DMD_VA_LOG")) + fprintf(stderr, "tva: contiguous NV12 dmabuf=%d size=%zu y=%ux%u " + "stride=%u uv_offset=%zu\n", dmabuf, total_size, + bridge_templ.width, bridge_templ.height, y_stride, uv_offset); + return result; + +fail: + for (unsigned i = 0; i < ARRAY_SIZE(resources); i++) + pipe_resource_reference(&resources[i], NULL); + close(dmabuf); + return NULL; +} +#endif + +/* KGSL exposes the same dma-buf through the VA producer and Chrome's ANGLE + * consumer, but it does not provide an implicit cross-context GPU dependency + * for a Gallium texture upload. Bracket CPU writes with the dma-buf exporter + * cache hooks so a consumer importing the fd observes completed frame data. */ +static int +tva_dmabuf_write_begin(struct pipe_context *pipe, struct pipe_resource *res) +{ + if (!pipe || !pipe->screen || !pipe->screen->resource_get_handle) + return -1; + + struct winsys_handle whandle; + memset(&whandle, 0, sizeof(whandle)); + whandle.type = WINSYS_HANDLE_TYPE_FD; + if (!pipe->screen->resource_get_handle(pipe->screen, pipe, res, + &whandle, + PIPE_HANDLE_USAGE_FRAMEBUFFER_WRITE)) + return -1; + + struct dma_buf_sync sync = { + .flags = DMA_BUF_SYNC_START | DMA_BUF_SYNC_WRITE, + }; + if (ioctl(whandle.handle, DMA_BUF_IOCTL_SYNC, &sync) < 0) { + int err = errno; + if (err != ENOTTY && err != EOPNOTSUPP && err != ENOSYS && + getenv("DMD_VA_LOG")) + fprintf(stderr, "tva: dma-buf write sync start failed fd=%d errno=%d\n", + whandle.handle, err); + close(whandle.handle); + return -1; + } + return whandle.handle; +} + +static void +tva_dmabuf_write_end(int fd) +{ + if (fd < 0) + return; + + struct dma_buf_sync sync = { + .flags = DMA_BUF_SYNC_END | DMA_BUF_SYNC_WRITE, + }; + if (ioctl(fd, DMA_BUF_IOCTL_SYNC, &sync) < 0 && getenv("DMD_VA_LOG")) + fprintf(stderr, "tva: dma-buf write sync end failed fd=%d errno=%d\n", + fd, errno); + close(fd); +} + +static bool +tva_reader_copy_enabled(void) +{ + const char *e = getenv("DMD_VA_READER_COPY"); + if (!e || !*e) + return true; + return !(!strcmp(e, "0") || !strcmp(e, "false") || + !strcmp(e, "off")); +} + +/* Copy directly into a linear dma-buf without touching pipe_context. The + * reader thread can therefore complete a surface as soon as the daemon + * output arrives, while the application-thread Gallium path remains available + * as a fallback for drivers that cannot mmap an exported BO. */ +static bool +tva_direct_copy_plane(struct pipe_screen *screen, struct pipe_resource *res, + const uint8_t *data, unsigned w, unsigned h, + unsigned src_stride) +{ + if (!screen || !screen->resource_get_handle || !res || !data || !w || !h) + return false; + + const unsigned blocksize = util_format_get_blocksize(res->format); + if (!blocksize || w > UINT_MAX / blocksize) + return false; + const size_t row_bytes = (size_t)w * blocksize; + if (src_stride < row_bytes) + return false; + + struct winsys_handle whandle; + memset(&whandle, 0, sizeof(whandle)); + whandle.type = WINSYS_HANDLE_TYPE_FD; + if (!screen->resource_get_handle(screen, NULL, res, &whandle, + PIPE_HANDLE_USAGE_FRAMEBUFFER_WRITE)) + return false; + + const int fd = whandle.handle; + const size_t dst_stride = whandle.stride; + const uint64_t object_size = whandle.size; + if (fd < 0 || dst_stride < row_bytes || !object_size || + (uint64_t)(h - 1) > (UINT64_MAX - row_bytes) / dst_stride || + (uint64_t)(h - 1) * dst_stride + row_bytes > object_size) { + if (fd >= 0) + close(fd); + return false; + } + + long page_size = sysconf(_SC_PAGESIZE); + if (page_size <= 0) + page_size = 4096; + const uint64_t page_mask = (uint64_t)page_size - 1; + const uint64_t map_offset = whandle.offset & ~page_mask; + const uint64_t delta = whandle.offset - map_offset; + const uint64_t need = (uint64_t)(h - 1) * dst_stride + row_bytes; + if (delta > object_size || need > object_size - delta || + delta + need > SIZE_MAX) { + close(fd); + return false; + } + + const size_t map_len = (size_t)(delta + need); + uint8_t *map = mmap(NULL, map_len, PROT_READ | PROT_WRITE, MAP_SHARED, + fd, (off_t)map_offset); + if (map == MAP_FAILED) { + close(fd); + return false; + } + + bool synced = false; + struct dma_buf_sync sync = { + .flags = DMA_BUF_SYNC_START | DMA_BUF_SYNC_WRITE, + }; + if (ioctl(fd, DMA_BUF_IOCTL_SYNC, &sync) == 0) { + synced = true; + } else if (errno != ENOTTY && errno != EOPNOTSUPP && errno != ENOSYS && + getenv("DMD_VA_LOG")) { + fprintf(stderr, "tva: direct dma-buf sync start failed fd=%d errno=%d\n", + fd, errno); + } + + uint8_t *dst = map + delta; + if (dst_stride == src_stride) { + memcpy(dst, data, row_bytes * h); + } else { + for (unsigned y = 0; y < h; y++) + memcpy(dst + (size_t)y * dst_stride, + data + (size_t)y * src_stride, row_bytes); + } + + if (synced) { + sync.flags = DMA_BUF_SYNC_END | DMA_BUF_SYNC_WRITE; + if (ioctl(fd, DMA_BUF_IOCTL_SYNC, &sync) < 0 && getenv("DMD_VA_LOG")) + fprintf(stderr, "tva: direct dma-buf sync end failed fd=%d errno=%d\n", + fd, errno); + } + munmap(map, map_len); + close(fd); + return true; +} + +static bool +tva_copy_frame_direct(struct tva_codec *c, struct tva_pending *p, + const uint8_t *src, size_t src_size) +{ + if (!c || !c->pipe || !p || !src || !p->resources[0] || + !p->resources[1] || !p->frame_width || !p->frame_height || + p->stride <= 0 || p->slice_height <= 0 || p->crop_left < 0 || + p->crop_top < 0 || p->crop_right < p->crop_left || + p->crop_bottom < p->crop_top || + p->crop_right >= (int)p->frame_width || + p->crop_bottom >= (int)p->frame_height || + p->crop_right >= p->stride || p->slice_height < (int)p->frame_height) + return false; + + const unsigned display_w = + (unsigned)(p->crop_right - p->crop_left + 1); + const unsigned display_h = + (unsigned)(p->crop_bottom - p->crop_top + 1); + const unsigned w = p->resources[0]->width0 < display_w + ? p->resources[0]->width0 : display_w; + const unsigned h = p->resources[0]->height0 < display_h + ? p->resources[0]->height0 : display_h; + if (!w || !h || p->resources[1]->width0 < (w + 1) / 2 || + p->resources[1]->height0 < (h + 1) / 2) + return false; + + const size_t stride = (size_t)p->stride; + const unsigned uv_w = (w + 1) / 2; + const unsigned uv_h = (h + 1) / 2; + const unsigned y_blocksize = + util_format_get_blocksize(p->resources[0]->format); + const unsigned uv_blocksize = + util_format_get_blocksize(p->resources[1]->format); + if (!y_blocksize || !uv_blocksize || w > UINT_MAX / y_blocksize || + uv_w > UINT_MAX / uv_blocksize || stride < (size_t)w * y_blocksize || + stride < (size_t)uv_w * uv_blocksize) + return false; + + if ((size_t)p->crop_top > SIZE_MAX / stride || + (size_t)p->crop_top * stride > SIZE_MAX - (size_t)p->crop_left || + (size_t)p->slice_height > SIZE_MAX / stride || + (size_t)(p->crop_top / 2) > SIZE_MAX / stride) + return false; + const size_t y_offset = (size_t)p->crop_top * stride + + (size_t)p->crop_left; + size_t uv_offset = (size_t)p->slice_height * stride; + if (uv_offset > SIZE_MAX - (size_t)(p->crop_top / 2) * stride) + return false; + uv_offset += (size_t)(p->crop_top / 2) * stride; + if (uv_offset > SIZE_MAX - (size_t)(p->crop_left & ~1)) + return false; + uv_offset += (size_t)(p->crop_left & ~1); + + const size_t y_rows = h - 1; + const size_t uv_rows = uv_h - 1; + const size_t y_row_bytes = (size_t)w * y_blocksize; + const size_t uv_row_bytes = (size_t)uv_w * uv_blocksize; + if (y_rows > SIZE_MAX / stride || + y_offset > SIZE_MAX - y_rows * stride || + y_offset + y_rows * stride > SIZE_MAX - y_row_bytes || + uv_rows > SIZE_MAX / stride || + uv_offset > SIZE_MAX - uv_rows * stride || + uv_offset + uv_rows * stride > SIZE_MAX - uv_row_bytes) + return false; + const size_t y_end = y_offset + y_rows * stride + y_row_bytes; + const size_t uv_end = uv_offset + uv_rows * stride + uv_row_bytes; + if (y_end > src_size || uv_end > src_size) + return false; + + struct pipe_screen *screen = c->pipe->screen; + if (!tva_direct_copy_plane(screen, p->resources[0], src + y_offset, + w, h, (unsigned)p->stride)) + return false; + if (!tva_direct_copy_plane(screen, p->resources[1], src + uv_offset, + uv_w, uv_h, (unsigned)p->stride)) + return false; + return true; +} +#endif + +static bool +tva_copy_plane(struct pipe_context *pipe, struct pipe_resource *res, + const uint8_t *data, unsigned w, unsigned h, unsigned stride) +{ + unsigned blocksize = util_format_get_blocksize(res->format); + if (!blocksize || w > UINT_MAX / blocksize || + stride < w * blocksize) + return false; + + struct pipe_box box = { + .x = 0, + .width = (int)w, + .y = 0, + .height = (int)h, + .z = 0, + .depth = 1, + }; + TVA_TRACE("copy plane fmt=%s %ux%u stride=%u src=%02x %02x %02x %02x", + util_format_short_name(res->format), w, h, stride, + data[0], data[1], data[2], data[3]); + /* CPU-mapped uploads are used by default for the cache-safe KGSL handoff. + * Callers can force the asynchronous GPU upload path with + * DMD_VA_CPU_COPY=0. */ + if (tva_cpu_copy_enabled() && pipe->texture_map && + pipe->texture_unmap) { + TVA_TRACE("copy path=cpu"); +#ifndef _WIN32 + int sync_fd = tva_dmabuf_write_begin(pipe, res); +#else + int sync_fd = -1; +#endif + struct pipe_transfer *transfer = NULL; + uint8_t *dst = pipe->texture_map(pipe, res, 0, PIPE_MAP_WRITE, + &box, &transfer); + if (!dst || !transfer) { + if (transfer) + pipe->texture_unmap(pipe, transfer); +#ifndef _WIN32 + tva_dmabuf_write_end(sync_fd); +#endif + return false; + } + unsigned row_bytes = w * blocksize; + if (transfer->stride < row_bytes) { + pipe->texture_unmap(pipe, transfer); +#ifndef _WIN32 + tva_dmabuf_write_end(sync_fd); +#endif + return false; + } + /* Bridge surfaces are linear and normally preserve the daemon's + * pitch. Collapse the row loop to one transfer in that common case; + * AV1 at 1080p otherwise performs more than 1,600 tiny memcpys per + * frame (and quickly becomes CPU-bound at 60 fps). */ + if (transfer->stride == stride) { + memcpy(dst, data, (size_t)row_bytes * h); + } else { + for (unsigned y = 0; y < h; y++) + memcpy(dst + (size_t)y * transfer->stride, + data + (size_t)y * stride, row_bytes); + } + pipe->texture_unmap(pipe, transfer); +#ifndef _WIN32 + tva_dmabuf_write_end(sync_fd); +#endif + } else if (pipe->texture_subdata) { + TVA_TRACE("copy path=gpu"); + pipe->texture_subdata(pipe, res, 0, PIPE_MAP_WRITE, &box, data, + stride, (uintptr_t)stride); + } else { + return false; + } + return true; +} + +/* texture_subdata() is asynchronous when the real context is wrapped by + * threaded_context. The bridge fence is the completion point exposed to VA + * clients, so make the queued resource writes visible before signaling it. */ +static bool +tva_flush_copy(struct pipe_context *pipe) +{ + struct pipe_fence_handle *fence = NULL; + struct pipe_screen *screen; + bool ready = true; + + if (!pipe || !pipe->flush || !(screen = pipe->screen)) + return false; + + pipe->flush(pipe, &fence, 0); + if (fence) { + ready = screen->fence_finish(screen, pipe, fence, + OS_TIMEOUT_INFINITE); + screen->fence_reference(screen, &fence, NULL); + } + TVA_TRACE("copy flush ready=%d", ready); + return ready; +} + +static void +tva_probe_resource(struct pipe_context *pipe, struct pipe_resource *res, + unsigned width) +{ + const char *e = getenv("DMD_VA_PROBE"); + if (!e || e[0] != '1' || !pipe || !res || !pipe->texture_map || + !pipe->texture_unmap) + return; + + struct pipe_box box = { + .x = 0, + .width = (int)MIN2(width, 8u), + .y = 0, + .height = 1, + .z = 0, + .depth = 1, + }; + struct pipe_transfer *transfer = NULL; + uint8_t *map = pipe->texture_map(pipe, res, 0, PIPE_MAP_READ, &box, + &transfer); + if (!map || !transfer) { + fprintf(stderr, "tva: resource probe map failed res=%p\n", + (void *)res); + if (transfer) + pipe->texture_unmap(pipe, transfer); + return; + } + fprintf(stderr, "tva: resource probe res=%p stride=%u bytes=%02x %02x %02x %02x %02x %02x %02x %02x\n", + (void *)res, transfer->stride, map[0], map[1], map[2], map[3], + map[4], map[5], map[6], map[7]); + pipe->texture_unmap(pipe, transfer); +} + +static bool +tva_copy_frame(struct tva_codec *c, struct tva_pending *p) +{ + const uint64_t copy_start_ns = os_time_get_nano(); + if (!p->staging || !p->resources[0] || !p->resources[1] || + !p->frame_width || !p->frame_height || p->stride <= 0 || + p->slice_height <= 0 || p->crop_left < 0 || p->crop_top < 0 || + p->crop_right < p->crop_left || p->crop_bottom < p->crop_top || + p->crop_right >= (int)p->frame_width || + p->crop_bottom >= (int)p->frame_height || + p->crop_right >= p->stride || p->slice_height < (int)p->frame_height) + return false; + + unsigned display_w = (unsigned)(p->crop_right - p->crop_left + 1); + unsigned display_h = (unsigned)(p->crop_bottom - p->crop_top + 1); + unsigned w = p->resources[0]->width0 < display_w ? + p->resources[0]->width0 : display_w; + unsigned h = p->resources[0]->height0 < display_h ? + p->resources[0]->height0 : display_h; + if (!w || !h || p->resources[1]->width0 < (w + 1) / 2 || + p->resources[1]->height0 < (h + 1) / 2) + return false; + + size_t stride = (size_t)p->stride; + size_t y_offset, uv_offset, y_end, uv_end; + size_t y_rows = h - 1; + size_t uv_rows = (h + 1) / 2 - 1; + unsigned uv_w = (w + 1) / 2; + unsigned uv_h = (h + 1) / 2; + unsigned y_blocksize = util_format_get_blocksize(p->resources[0]->format); + unsigned uv_blocksize = util_format_get_blocksize(p->resources[1]->format); + size_t y_row_bytes, uv_row_bytes; + if (!y_blocksize || !uv_blocksize || + w > UINT_MAX / y_blocksize || uv_w > UINT_MAX / uv_blocksize) + return false; + y_row_bytes = (size_t)w * y_blocksize; + uv_row_bytes = (size_t)uv_w * uv_blocksize; + if (stride < y_row_bytes || stride < uv_row_bytes || + p->resources[1]->width0 < uv_w || p->resources[1]->height0 < uv_h) + return false; + if ((size_t)p->crop_top > SIZE_MAX / stride || + (size_t)p->crop_top * stride > SIZE_MAX - (size_t)p->crop_left) + return false; + y_offset = (size_t)p->crop_top * stride + (size_t)p->crop_left; + if ((size_t)p->slice_height > SIZE_MAX / stride) + return false; + uv_offset = (size_t)p->slice_height * stride; + if ((size_t)(p->crop_top / 2) > SIZE_MAX / stride || + uv_offset > SIZE_MAX - (size_t)(p->crop_top / 2) * stride) + return false; + uv_offset += (size_t)(p->crop_top / 2) * stride; + if (uv_offset > SIZE_MAX - (size_t)(p->crop_left & ~1)) + return false; + uv_offset += (size_t)(p->crop_left & ~1); + if (y_rows > SIZE_MAX / stride || + y_offset > SIZE_MAX - y_rows * stride || + y_offset + y_rows * stride > SIZE_MAX - y_row_bytes) + return false; + y_end = y_offset + y_rows * stride + y_row_bytes; + if (uv_rows > SIZE_MAX / stride || + uv_offset > SIZE_MAX - uv_rows * stride || + uv_offset + uv_rows * stride > SIZE_MAX - uv_row_bytes) + return false; + uv_end = uv_offset + uv_rows * stride + uv_row_bytes; + if (y_end > p->staging_size || uv_end > p->staging_size) + return false; + + struct pipe_context *pipe = c->pipe; + if (!pipe) + return false; + + /* CPU mappings write the exported allocation synchronously; there is no + * Gallium batch to submit in that path. Keep the information here so the + * caller can avoid flushing unrelated GPU work after every decoded frame. + * The GPU-upload path still needs an explicit flush before its dma-buf is + * handed to the consumer. */ + const bool cpu_copy = tva_cpu_copy_enabled() && pipe->texture_map && + pipe->texture_unmap; + + TVA_TRACE("copy frame unit=%u frame=%ux%u stride=%d slice=%d crop=%d,%d-%d,%d yoff=%zu uvoff=%zu size=%zu", + p->unit_seq, p->frame_width, p->frame_height, p->stride, + p->slice_height, p->crop_left, p->crop_top, p->crop_right, + p->crop_bottom, y_offset, uv_offset, p->staging_size); + + if (!tva_copy_plane(pipe, p->resources[0], p->staging + y_offset, + w, h, (unsigned)p->stride)) + return false; + if (!tva_copy_plane(pipe, p->resources[1], p->staging + uv_offset, + uv_w, uv_h, (unsigned)p->stride)) + return false; + if (!cpu_copy && !tva_flush_copy(pipe)) + return false; + + tva_probe_resource(pipe, p->resources[0], w); + tva_probe_resource(pipe, p->resources[1], uv_w * 2); + TVA_TRACE("copy frame complete unit=%u duration=%.3f ms", + p->unit_seq, + (double)(os_time_get_nano() - copy_start_ns) / 1000000.0); + return true; +} + +/* ---------------------------- reader thread */ +/* + * The reader thread moves frames from the session into pending staging + * buffers. It exists because the two socket directions must always flow: + * the application thread sends AUs inside EndPicture while MediaCodec's + * input buffers are throttled by output consumption - with a single + * thread, a blocking send would stall reads and the C2 decoder's input + * queue would deadlock the pipeline. + * + * The reader only touches the session socket, the pending ring (under + * pend_mutex) and its own staging buffers. pipe_context stays on the + * application thread: the surface copy happens in fence_wait. + */ +static int +tva_reader_thread(void *param) +{ + struct tva_codec *c = param; + + for (;;) { + mtx_lock(&c->pend_mutex); + bool quitting = c->quitting; + mtx_unlock(&c->pend_mutex); + if (quitting) + break; + + struct tva_frame f; + int r = tva_session_next_frame(c->sess, &f, 200); + if (r == TVA_ERR_TIMEOUT) + continue; + if (r == TVA_EOS) { + mtx_lock(&c->pend_mutex); + tva_mark_broken_locked(c); + mtx_unlock(&c->pend_mutex); + break; + } + if (r < 0) { + mtx_lock(&c->pend_mutex); + tva_mark_broken_locked(c); + mtx_unlock(&c->pend_mutex); + break; + } + + TVA_TRACE("reader frame unit=%u size=%zu slot=%d", f.unit_seq, + f.size, f.shm_slot); + if (getenv("DMD_VA_PROBE") && f.data && f.size >= 1920u * 1088u) { + size_t center = 540u * 1920u + 960u; + fprintf(stderr, "tva: reader probe unit=%u y0=%02x ycenter=%02x uv0=%02x %02x\n", + f.unit_seq, f.data[0], f.data[center], + f.data[1920u * 1088u], f.data[1920u * 1088u + 1]); + } + + /* Match the frame to a pending picture by unit index. Unknown + * indices fall back to the oldest waiting entry for old peers. */ + mtx_lock(&c->pend_mutex); + struct tva_pending *p = tva_pend_find_locked(c, f.unit_seq); + if (!p || p->ready) { + mtx_unlock(&c->pend_mutex); + tva_session_release_frame(c->sess, &f); + continue; + } + p->frame_width = f.width; + p->frame_height = f.height; + p->stride = f.stride; + p->slice_height = f.slice_height; + p->crop_left = f.crop_left; + p->crop_top = f.crop_top; + p->crop_right = f.crop_right; + p->crop_bottom = f.crop_bottom; + +#ifndef _WIN32 + /* A direct dma-buf copy closes the surface-reuse window for AV1 + * hidden references. It is enabled by default for the validated + * linear resources; DMD_VA_READER_COPY=0 opts out. A failed direct + * map falls back to the staged application-thread copy below. */ + if (tva_reader_copy_enabled() && tva_cpu_copy_enabled() && + tva_copy_frame_direct(c, p, f.data, f.size)) { + p->copied = true; + p->ready = true; + c->frames_done++; + TVA_TRACE("reader direct copy unit=%u result=1", p->unit_seq); + u_cnd_monotonic_broadcast(&c->pend_cond); + mtx_unlock(&c->pend_mutex); + tva_session_release_frame(c->sess, &f); + continue; + } +#endif + + p->staging = malloc(f.size ? f.size : 1); + if (!p->staging) { + tva_mark_broken_locked(c); + mtx_unlock(&c->pend_mutex); + tva_session_release_frame(c->sess, &f); + break; + } + memcpy(p->staging, f.data, f.size); + p->staging_size = f.size; + p->ready = true; + c->frames_done++; + u_cnd_monotonic_broadcast(&c->pend_cond); + mtx_unlock(&c->pend_mutex); + tva_session_release_frame(c->sess, &f); /* return the slot promptly */ + } + return 0; +} + +/* ---------------------------------- H.264 CSD synthesis (SPS/PPS) */ +/* + * FFmpeg's VAAPI H.264 path does not deliver SPS/PPS as slice-data buffers; + * the VA frontend exposes only the parsed pipe_h264_sps/pps structures. The + * daemon's MediaCodec needs the parameter sets as CSD, so the bridge + * regenerates them from those parsed structures. (Same role as upstream + * vaapi-driver's h264_bitstream.c, ported to the gallium-side data model.) + */ + +struct tva_bw { + uint8_t *buf; + size_t cap; + size_t len; /* bytes flushed to buf (EBSP) */ + uint32_t acc; + unsigned nbits; + unsigned zeros; /* running zero count for emulation prevention */ +}; + +static void +tva_bw_put(struct tva_bw *w, unsigned n, uint32_t v) +{ + for (int i = n - 1; i >= 0; i--) { + unsigned bit = (v >> i) & 1; + w->acc = (w->acc << 1) | bit; + if (++w->nbits == 8) { + uint8_t byte = (uint8_t)(w->acc & 0xff); + if (w->len + 4 > w->cap) { + w->cap = w->cap ? w->cap * 2 : 64; + w->buf = realloc(w->buf, w->cap); + } + /* emulation prevention: 00 00 {0,1,2,3} -> 00 00 03 xx */ + if (w->zeros >= 2 && byte <= 3) { + w->buf[w->len++] = 3; + w->zeros = 0; + } + w->buf[w->len++] = byte; + w->zeros = byte == 0 ? w->zeros + 1 : 0; + w->nbits = 0; + w->acc = 0; + } + } +} + +static void +tva_bw_ue(struct tva_bw *w, uint32_t v) +{ + uint32_t val = v + 1; + unsigned n = 0; + while ((val >> n) != 1) + n++; + tva_bw_put(w, n, 0); + tva_bw_put(w, n + 1, val); +} + +static void +tva_bw_se(struct tva_bw *w, int32_t v) +{ + uint32_t k = v <= 0 ? (uint32_t)(-2 * v) : (uint32_t)(2 * v - 1); + tva_bw_ue(w, k); +} + +static void +tva_bw_rbsp_trailing(struct tva_bw *w) +{ + tva_bw_put(w, 1, 1); + while (w->nbits) + tva_bw_put(w, 1, 0); +} + +static bool +tva_h264_high_profile(enum pipe_video_profile profile) +{ + return profile == PIPE_VIDEO_PROFILE_MPEG4_AVC_HIGH || + profile == PIPE_VIDEO_PROFILE_MPEG4_AVC_HIGH10 || + profile == PIPE_VIDEO_PROFILE_MPEG4_AVC_HIGH422 || + profile == PIPE_VIDEO_PROFILE_MPEG4_AVC_HIGH444; +} + +/* + * Build the SPS NALU (with NAL header + emulation prevention) from the + * frontend-parsed struct. Returns the RBSP size; the NAL header byte is + * written first (nal_ref_idc=3, type=7). + */ +static size_t +tva_build_h264_sps(enum pipe_video_profile profile, + const struct pipe_h264_sps *sps, unsigned max_refs, + unsigned visible_width, unsigned visible_height, + uint8_t **out) +{ + struct tva_bw w = {0}; + + /* VA-API carries the H.264 profile in the decode context rather than in + * VAPictureParameterBufferH264. Preserve that profile in the synthetic + * SPS; inferring High solely from chroma/depth mislabels ordinary 8-bit + * 4:2:0 High streams as Main. */ + uint8_t profile_idc; + switch (profile) { + case PIPE_VIDEO_PROFILE_MPEG4_AVC_BASELINE: + case PIPE_VIDEO_PROFILE_MPEG4_AVC_CONSTRAINED_BASELINE: + profile_idc = 66; + break; + case PIPE_VIDEO_PROFILE_MPEG4_AVC_MAIN: + profile_idc = 77; + break; + case PIPE_VIDEO_PROFILE_MPEG4_AVC_EXTENDED: + profile_idc = 88; + break; + case PIPE_VIDEO_PROFILE_MPEG4_AVC_HIGH: + profile_idc = 100; + break; + case PIPE_VIDEO_PROFILE_MPEG4_AVC_HIGH10: + profile_idc = 110; + break; + case PIPE_VIDEO_PROFILE_MPEG4_AVC_HIGH422: + profile_idc = 122; + break; + case PIPE_VIDEO_PROFILE_MPEG4_AVC_HIGH444: + profile_idc = 244; + break; + default: + /* Keep a conservative fallback for callers that pass an unknown + * profile while still providing high-profile SPS fields when the + * parsed descriptor requires them. */ + profile_idc = (sps->bit_depth_luma_minus8 || + sps->bit_depth_chroma_minus8 || + sps->chroma_format_idc > 1) ? 100 : 77; + break; + } + + tva_bw_put(&w, 8, 0x67); /* nal_ref_idc=3, type=7 */ + tva_bw_put(&w, 8, profile_idc); + tva_bw_put(&w, 8, 0); /* constraint flags + reserved */ + tva_bw_put(&w, 8, sps->level_idc ? sps->level_idc : 40); + tva_bw_ue(&w, 0); /* seq_parameter_set_id */ + if (profile_idc == 100 || profile_idc == 110 || + profile_idc == 122 || profile_idc == 244) { + tva_bw_ue(&w, sps->chroma_format_idc); + if (sps->chroma_format_idc == 3) + tva_bw_put(&w, 1, sps->separate_colour_plane_flag); + tva_bw_ue(&w, sps->bit_depth_luma_minus8); + tva_bw_ue(&w, sps->bit_depth_chroma_minus8); + tva_bw_put(&w, 1, 0); /* qpprime_y_zero_transform_bypass */ + tva_bw_put(&w, 1, 0); /* seq_scaling_matrix_present */ + } + tva_bw_ue(&w, sps->log2_max_frame_num_minus4); + tva_bw_ue(&w, sps->pic_order_cnt_type); + if (sps->pic_order_cnt_type == 0) + tva_bw_ue(&w, sps->log2_max_pic_order_cnt_lsb_minus4); + else if (sps->pic_order_cnt_type == 1) { + tva_bw_put(&w, 1, sps->delta_pic_order_always_zero_flag); + tva_bw_se(&w, sps->offset_for_non_ref_pic); + tva_bw_se(&w, sps->offset_for_top_to_bottom_field); + tva_bw_ue(&w, sps->num_ref_frames_in_pic_order_cnt_cycle); + for (unsigned i = 0; i < sps->num_ref_frames_in_pic_order_cnt_cycle; i++) + tva_bw_se(&w, sps->offset_for_ref_frame[i]); + } + /* the VA picture param carries the real DPB size; sps->max_num_ref_frames + * itself is never filled by the frontend */ + tva_bw_ue(&w, max_refs ? max_refs : 1); + tva_bw_put(&w, 1, sps->gaps_in_frame_num_value_allowed_flag); + tva_bw_ue(&w, sps->pic_width_in_mbs_minus1); + tva_bw_ue(&w, sps->pic_height_in_mbs_minus1); + tva_bw_put(&w, 1, sps->frame_mbs_only_flag); + if (!sps->frame_mbs_only_flag) + tva_bw_put(&w, 1, sps->mb_adaptive_frame_field_flag); + tva_bw_put(&w, 1, sps->direct_8x8_inference_flag); + + /* VA exposes the coded macroblock dimensions, while the video template + * retains the visible dimensions requested by the application. Preserve + * a right/bottom crop when those differ; Qualcomm's decoder otherwise + * treats the synthetic stream as 816 pixels wide and may reject the + * 810-pixel output surfaces. */ + unsigned coded_width = (sps->pic_width_in_mbs_minus1 + 1) * 16; + unsigned coded_height = (sps->pic_height_in_mbs_minus1 + 1) * 16 * + (sps->frame_mbs_only_flag ? 1 : 2); + unsigned crop_unit_x = sps->chroma_format_idc == 0 ? 1 : 2; + unsigned crop_unit_y = sps->chroma_format_idc == 0 ? + (sps->frame_mbs_only_flag ? 1 : 2) : + (sps->frame_mbs_only_flag ? 2 : 4); + unsigned crop_right = 0, crop_bottom = 0; + if (visible_width && coded_width > visible_width && + (coded_width - visible_width) % crop_unit_x == 0) + crop_right = (coded_width - visible_width) / crop_unit_x; + if (visible_height && coded_height > visible_height && + (coded_height - visible_height) % crop_unit_y == 0) + crop_bottom = (coded_height - visible_height) / crop_unit_y; + bool cropped = crop_right || crop_bottom; + tva_bw_put(&w, 1, cropped); + if (cropped) { + tva_bw_ue(&w, 0); /* frame_crop_left_offset */ + tva_bw_ue(&w, crop_right); + tva_bw_ue(&w, 0); /* frame_crop_top_offset */ + tva_bw_ue(&w, crop_bottom); + } + /* The VA picture descriptor does not carry the original VUI. Emit only a + * minimal parameter block and, importantly, do not invent a bitstream + * restriction: the source stream may permit reordering (the test stream + * does), and Qualcomm C2 treats a fabricated max_num_reorder_frames=0 as + * a different stream, dropping output after its initial reorder window. + * The remaining VUI flags are left absent because their values are not + * represented by the VA descriptor. */ + tva_bw_put(&w, 1, 1); /* vui_parameters_present */ + tva_bw_put(&w, 1, 0); /* aspect_ratio_info_present */ + tva_bw_put(&w, 1, 0); /* overscan_info_present */ + tva_bw_put(&w, 1, 0); /* video_signal_type_present */ + tva_bw_put(&w, 1, 0); /* chroma_loc_info_present */ + tva_bw_put(&w, 1, 0); /* timing_info_present */ + tva_bw_put(&w, 1, 0); /* nal_hrd_parameters_present */ + tva_bw_put(&w, 1, 0); /* vcl_hrd_parameters_present */ + tva_bw_put(&w, 1, 0); /* pic_struct_present */ + tva_bw_put(&w, 1, 0); /* bitstream_restriction_flag */ + tva_bw_rbsp_trailing(&w); + *out = w.buf; + return w.len; +} + +/* PPS NALU: nal_ref_idc=3, type=8 */ +static size_t +tva_build_h264_pps(enum pipe_video_profile profile, + const struct pipe_h264_pps *pps, + unsigned default_l0, unsigned default_l1, + uint8_t **out) +{ + struct tva_bw w = {0}; + + tva_bw_put(&w, 8, 0x68); + tva_bw_ue(&w, 0); /* pic_parameter_set_id */ + tva_bw_ue(&w, 0); /* seq_parameter_set_id */ + tva_bw_put(&w, 1, pps->entropy_coding_mode_flag); + tva_bw_put(&w, 1, pps->bottom_field_pic_order_in_frame_present_flag); + tva_bw_ue(&w, pps->num_slice_groups_minus1); + /* VAPictureParameterBufferH264 has no PPS default-reference fields. The + * caller supplies defaults inferred from the slice parameters (or the + * decoder DPB size when slice parameters are unavailable). */ + bool high_profile = tva_h264_high_profile(profile); + tva_bw_ue(&w, pps->num_ref_idx_l0_default_active_minus1 ? + pps->num_ref_idx_l0_default_active_minus1 : default_l0); + tva_bw_ue(&w, pps->num_ref_idx_l1_default_active_minus1 ? + pps->num_ref_idx_l1_default_active_minus1 : default_l1); + tva_bw_put(&w, 1, pps->weighted_pred_flag); + tva_bw_put(&w, 2, pps->weighted_bipred_idc); + tva_bw_se(&w, pps->pic_init_qp_minus26); + tva_bw_se(&w, pps->pic_init_qs_minus26); + tva_bw_se(&w, pps->chroma_qp_index_offset); + tva_bw_put(&w, 1, pps->deblocking_filter_control_present_flag); + tva_bw_put(&w, 1, pps->constrained_intra_pred_flag); + tva_bw_put(&w, 1, pps->redundant_pic_cnt_present_flag); + if (high_profile) { + tva_bw_put(&w, 1, pps->transform_8x8_mode_flag); + tva_bw_put(&w, 1, 0); /* pic_scaling_matrix_present */ + tva_bw_se(&w, pps->second_chroma_qp_index_offset); + } + tva_bw_rbsp_trailing(&w); + *out = w.buf; + return w.len; +} + +/* ---------------------------------- HEVC CSD synthesis (VPS/SPS/PPS) */ +/* + * VA-API gives the frontend parsed HEVC picture parameters, but does not + * carry the original VPS/SPS/PPS byte stream. MediaCodec needs those + * parameter sets before the first VCL unit, so reconstruct the minimal + * single-layer Annex B headers from the pipe_h265 descriptors. Slice data + * itself is forwarded unchanged: VA's slice-data offsets guarantee that the + * slice header is still present in the buffer. + * + * A few syntax elements are not represented by the VA decode descriptor. In + * particular, the contents of SPS short-term reference-picture sets cannot be + * reconstructed from the VA descriptor. The Qualcomm decoder still requires + * the declared RPS count to be non-zero, even when every slice carries its own + * inline RPS, so the bridge emits empty placeholder sets and rejects a slice + * that instead references an SPS RPS. + */ + +static unsigned +tva_h265_profile_idc(enum pipe_video_profile profile) +{ + return profile == PIPE_VIDEO_PROFILE_HEVC_MAIN_10 ? 2 : 1; +} + +/* Return a conservative level_idc based on the luma-picture size. VA-API's + * HEVC picture descriptor does not expose the stream's original level. */ +static unsigned +tva_h265_level_idc(unsigned width, unsigned height) +{ + uint64_t luma = (uint64_t)width * height; + if (luma <= 36864) return 30; /* 1.0 */ + if (luma <= 122880) return 60; /* 2.0 */ + if (luma <= 245760) return 63; /* 2.1 */ + if (luma <= 552960) return 90; /* 3.0 */ + if (luma <= 983040) return 93; /* 3.1 */ + if (luma <= 2228224) return 120; /* 4.0 */ + if (luma <= 8912896) return 150; /* 5.0 */ + return 180; /* 6.0 */ +} + +static void +tva_bw_hevc_header(struct tva_bw *w, unsigned nal_type) +{ + /* forbidden_zero_bit=0, nal_unit_type, nuh_layer_id=0, + * nuh_temporal_id_plus1=1. */ + tva_bw_put(w, 8, (nal_type & 0x3f) << 1); + tva_bw_put(w, 8, 1); +} + +static void +tva_h265_put_ptl(struct tva_bw *w, enum pipe_video_profile profile, + unsigned width, unsigned height) +{ + unsigned profile_idc = tva_h265_profile_idc(profile); + + tva_bw_put(w, 2, 0); /* general_profile_space */ + tva_bw_put(w, 1, 0); /* general_tier_flag */ + tva_bw_put(w, 5, profile_idc); /* general_profile_idc */ + for (unsigned i = 0; i < 32; i++) + tva_bw_put(w, 1, i == profile_idc); + tva_bw_put(w, 1, 1); /* progressive_source */ + tva_bw_put(w, 1, 0); /* interlaced_source */ + tva_bw_put(w, 1, 0); /* non_packed_constraint */ + tva_bw_put(w, 1, 1); /* frame_only_constraint */ + tva_bw_put(w, 22, 0); /* reserved_zero_43bits (part 1) */ + tva_bw_put(w, 21, 0); /* reserved_zero_43bits (part 2) */ + tva_bw_put(w, 1, 0); /* inbld/reserved_zero_bit */ + tva_bw_put(w, 8, tva_h265_level_idc(width, height)); +} + +static size_t +tva_build_h265_vps(const struct pipe_h265_sps *sps, + enum pipe_video_profile profile, uint8_t **out) +{ + struct tva_bw w = {0}; + + tva_bw_hevc_header(&w, 32); + tva_bw_put(&w, 4, 0); /* vps_video_parameter_set_id */ + tva_bw_put(&w, 2, 3); /* base_layer_internal/available */ + tva_bw_put(&w, 6, 0); /* vps_max_layers_minus1 */ + tva_bw_put(&w, 3, 0); /* vps_max_sub_layers_minus1 */ + tva_bw_put(&w, 1, 1); /* vps_temporal_id_nesting_flag */ + tva_bw_put(&w, 16, 0xffff); /* vps_reserved_0xffff_16bits */ + tva_h265_put_ptl(&w, profile, sps->pic_width_in_luma_samples, + sps->pic_height_in_luma_samples); + tva_bw_put(&w, 1, 0); /* sub_layer_ordering_info_present */ + tva_bw_ue(&w, sps->sps_max_dec_pic_buffering_minus1); + tva_bw_ue(&w, 0); /* vps_max_num_reorder_pics */ + tva_bw_ue(&w, 0); /* vps_max_latency_increase_plus1 */ + tva_bw_put(&w, 6, 0); /* vps_max_layer_id */ + tva_bw_ue(&w, 0); /* vps_num_layer_sets_minus1 */ + tva_bw_put(&w, 1, 0); /* vps_timing_info_present_flag */ + tva_bw_put(&w, 1, 0); /* vps_extension_flag */ + tva_bw_rbsp_trailing(&w); + *out = w.buf; + return w.len; +} + +static size_t +tva_build_h265_sps(const struct pipe_h265_sps *sps, + enum pipe_video_profile profile, uint8_t **out) +{ + struct tva_bw w = {0}; + + tva_bw_hevc_header(&w, 33); + tva_bw_put(&w, 4, 0); /* sps_video_parameter_set_id */ + tva_bw_put(&w, 3, 0); /* sps_max_sub_layers_minus1 */ + tva_bw_put(&w, 1, 1); /* sps_temporal_id_nesting_flag */ + tva_h265_put_ptl(&w, profile, sps->pic_width_in_luma_samples, + sps->pic_height_in_luma_samples); + tva_bw_ue(&w, 0); /* sps_seq_parameter_set_id */ + tva_bw_ue(&w, sps->chroma_format_idc); + if (sps->chroma_format_idc == 3) + tva_bw_put(&w, 1, sps->separate_colour_plane_flag); + tva_bw_ue(&w, sps->pic_width_in_luma_samples); + tva_bw_ue(&w, sps->pic_height_in_luma_samples); + tva_bw_put(&w, 1, 0); /* conformance_window_flag */ + tva_bw_ue(&w, sps->bit_depth_luma_minus8); + tva_bw_ue(&w, sps->bit_depth_chroma_minus8); + tva_bw_ue(&w, sps->log2_max_pic_order_cnt_lsb_minus4); + tva_bw_put(&w, 1, 0); /* sub_layer_ordering_info_present */ + tva_bw_ue(&w, sps->sps_max_dec_pic_buffering_minus1); + tva_bw_ue(&w, 0); /* sps_max_num_reorder_pics */ + tva_bw_ue(&w, 0); /* sps_max_latency_increase_plus1 */ + tva_bw_ue(&w, sps->log2_min_luma_coding_block_size_minus3); + tva_bw_ue(&w, sps->log2_diff_max_min_luma_coding_block_size); + tva_bw_ue(&w, sps->log2_min_transform_block_size_minus2); + tva_bw_ue(&w, sps->log2_diff_max_min_transform_block_size); + tva_bw_ue(&w, sps->max_transform_hierarchy_depth_inter); + tva_bw_ue(&w, sps->max_transform_hierarchy_depth_intra); + tva_bw_put(&w, 1, sps->scaling_list_enabled_flag); + if (sps->scaling_list_enabled_flag) + tva_bw_put(&w, 1, 0); /* use default scaling lists */ + tva_bw_put(&w, 1, sps->amp_enabled_flag); + tva_bw_put(&w, 1, sps->sample_adaptive_offset_enabled_flag); + tva_bw_put(&w, 1, sps->pcm_enabled_flag); + if (sps->pcm_enabled_flag) { + tva_bw_put(&w, 4, sps->pcm_sample_bit_depth_luma_minus1); + tva_bw_put(&w, 4, sps->pcm_sample_bit_depth_chroma_minus1); + tva_bw_ue(&w, sps->log2_min_pcm_luma_coding_block_size_minus3); + tva_bw_ue(&w, sps->log2_diff_max_min_luma_coding_block_size); + tva_bw_put(&w, 1, sps->pcm_loop_filter_disabled_flag); + } + /* Keep the count from VA-API. Its descriptor does not carry the RPS + * contents, so emit empty non-predicted sets as placeholders. The tested + * Qualcomm decoder requires a non-zero count even for inline slice RPS. + * tva_h265_can_build() rejects streams whose slices reference these + * placeholders through short_term_ref_pic_set_sps_flag. */ + tva_bw_ue(&w, sps->num_short_term_ref_pic_sets); + for (unsigned i = 0; i < sps->num_short_term_ref_pic_sets; i++) { + if (i) + tva_bw_put(&w, 1, 0); /* inter_ref_pic_set_prediction */ + tva_bw_ue(&w, 0); /* num_negative_pics */ + tva_bw_ue(&w, 0); /* num_positive_pics */ + } + tva_bw_put(&w, 1, sps->long_term_ref_pics_present_flag); + if (sps->long_term_ref_pics_present_flag) + tva_bw_ue(&w, sps->num_long_term_ref_pics_sps); + tva_bw_put(&w, 1, sps->sps_temporal_mvp_enabled_flag); + tva_bw_put(&w, 1, sps->strong_intra_smoothing_enabled_flag); + tva_bw_put(&w, 1, 0); /* vui_parameters_present_flag */ + tva_bw_put(&w, 1, 0); /* sps_extension_present_flag */ + tva_bw_rbsp_trailing(&w); + *out = w.buf; + return w.len; +} + +static size_t +tva_build_h265_pps(const struct pipe_h265_pps *pps, uint8_t **out) +{ + struct tva_bw w = {0}; + + if (pps->tiles_enabled_flag && + (pps->num_tile_columns_minus1 >= ARRAY_SIZE(pps->column_width_minus1) || + pps->num_tile_rows_minus1 >= ARRAY_SIZE(pps->row_height_minus1))) + return 0; + + tva_bw_hevc_header(&w, 34); + tva_bw_ue(&w, 0); /* pps_pic_parameter_set_id */ + tva_bw_ue(&w, 0); /* pps_seq_parameter_set_id */ + tva_bw_put(&w, 1, pps->dependent_slice_segments_enabled_flag); + tva_bw_put(&w, 1, pps->output_flag_present_flag); + tva_bw_put(&w, 3, pps->num_extra_slice_header_bits); + tva_bw_put(&w, 1, pps->sign_data_hiding_enabled_flag); + tva_bw_put(&w, 1, pps->cabac_init_present_flag); + tva_bw_ue(&w, pps->num_ref_idx_l0_default_active_minus1); + tva_bw_ue(&w, pps->num_ref_idx_l1_default_active_minus1); + tva_bw_se(&w, pps->init_qp_minus26); + tva_bw_put(&w, 1, pps->constrained_intra_pred_flag); + tva_bw_put(&w, 1, pps->transform_skip_enabled_flag); + tva_bw_put(&w, 1, pps->cu_qp_delta_enabled_flag); + if (pps->cu_qp_delta_enabled_flag) + tva_bw_ue(&w, pps->diff_cu_qp_delta_depth); + tva_bw_se(&w, pps->pps_cb_qp_offset); + tva_bw_se(&w, pps->pps_cr_qp_offset); + tva_bw_put(&w, 1, pps->pps_slice_chroma_qp_offsets_present_flag); + tva_bw_put(&w, 1, pps->weighted_pred_flag); + tva_bw_put(&w, 1, pps->weighted_bipred_flag); + tva_bw_put(&w, 1, pps->transquant_bypass_enabled_flag); + tva_bw_put(&w, 1, pps->tiles_enabled_flag); + tva_bw_put(&w, 1, pps->entropy_coding_sync_enabled_flag); + if (pps->tiles_enabled_flag) { + tva_bw_ue(&w, pps->num_tile_columns_minus1); + tva_bw_ue(&w, pps->num_tile_rows_minus1); + /* The frontend does not preserve uniform_spacing_flag. Explicit + * widths/heights are available, so use the non-uniform form. */ + tva_bw_put(&w, 1, 0); + for (unsigned i = 0; i < pps->num_tile_columns_minus1; i++) + tva_bw_ue(&w, pps->column_width_minus1[i]); + for (unsigned i = 0; i < pps->num_tile_rows_minus1; i++) + tva_bw_ue(&w, pps->row_height_minus1[i]); + tva_bw_put(&w, 1, pps->loop_filter_across_tiles_enabled_flag); + } + tva_bw_put(&w, 1, pps->pps_loop_filter_across_slices_enabled_flag); + + /* VA-API does not expose deblocking_filter_control_present_flag itself; + * infer it from the controls which follow it in the bitstream. */ + bool dbf_ctrl = pps->deblocking_filter_override_enabled_flag || + pps->pps_deblocking_filter_disabled_flag || + pps->pps_beta_offset_div2 || pps->pps_tc_offset_div2; + tva_bw_put(&w, 1, dbf_ctrl); + if (dbf_ctrl) { + tva_bw_put(&w, 1, pps->deblocking_filter_override_enabled_flag); + tva_bw_put(&w, 1, pps->pps_deblocking_filter_disabled_flag); + if (!pps->pps_deblocking_filter_disabled_flag) { + tva_bw_se(&w, pps->pps_beta_offset_div2); + tva_bw_se(&w, pps->pps_tc_offset_div2); + } + } + tva_bw_put(&w, 1, 0); /* pps_scaling_list_data_present */ + tva_bw_put(&w, 1, pps->lists_modification_present_flag); + tva_bw_ue(&w, pps->log2_parallel_merge_level_minus2); + tva_bw_put(&w, 1, pps->slice_segment_header_extension_present_flag); + tva_bw_put(&w, 1, 0); /* pps_extension_present_flag */ + tva_bw_rbsp_trailing(&w); + *out = w.buf; + return w.len; +} + +static bool +tva_h265_can_build(const struct pipe_h265_sps *sps, + const struct pipe_h265_picture_desc *pic) +{ + if (!sps || !pic) + return false; + /* VA-API exposes only the number of short-term RPS entries, not their + * contents. The slice header bit count is non-zero when the current + * picture carries an inline RPS (st_rps_bits in VAPictureParameterBuffer + * HEVC). An IDR picture has no RPS syntax, so it is also safe to seed the + * synthetic SPS there. If a later picture references an SPS RPS, the + * synthetic SPS cannot represent it and must be rejected. */ + if (sps->num_short_term_ref_pic_sets > 0 && + !pic->IDRPicFlag && pic->NumShortTermPictureSliceHeaderBits == 0) + return false; + /* Long-term SPS entries have the same limitation. */ + if (sps->long_term_ref_pics_present_flag && + sps->num_long_term_ref_pics_sps > 0) + return false; + return true; +} + +/* ---------------------------- codec vfuncs */ +static void +tva_codec_destroy_fence(struct pipe_video_codec *codec, + struct pipe_fence_handle *fence_handle); + +static void +tva_codec_begin_frame(struct pipe_video_codec *codec, + struct pipe_video_buffer *target, + struct pipe_picture_desc *picture) +{ + /* begin_frame is emitted once before the first slice data by the + * frontend; the bridge delimits pictures via end_frame instead. */ + (void)codec; (void)target; (void)picture; +} + +static void +tva_codec_decode_bitstream(struct pipe_video_codec *codec, + struct pipe_video_buffer *target, + struct pipe_picture_desc *picture, + unsigned num_buffers, + const void *const *buffers, + const unsigned *sizes) +{ + struct tva_codec *c = tva_codec(codec); + + (void)target; + (void)picture; + + /* The frontend unmaps the VA buffers after RenderPicture returns, so + * the data must be copied now. The buffers already carry H.264/HEVC + * start codes (the frontend prepends them when missing). */ + for (unsigned i = 0; i < num_buffers; i++) { + if (c->acc_len + sizes[i] > c->acc_cap) { + size_t nc = c->acc_cap ? c->acc_cap : 256 * 1024; + while (nc < c->acc_len + sizes[i]) + nc += nc / 2; + uint8_t *na = realloc(c->acc, nc); + if (!na) { + tva_mark_broken(c); + return; + } + c->acc = na; + c->acc_cap = nc; + } + memcpy(c->acc + c->acc_len, buffers[i], sizes[i]); + c->acc_len += sizes[i]; + } +} + +/* + * The Gallium AV1 descriptor is a compact, driver-facing representation of + * the VA-API picture parameters. AV1 OBU reconstruction uses the public VA + * structure because its field names and semantics are defined by libva, so + * copy the fields needed by the bitstream writer here. Fields that the + * Gallium frontend intentionally does not retain (color range and still + * picture) use the safe defaults for ordinary 8-bit 4:2:0 video. + */ +static bool +tva_av1_to_va_picture(const struct pipe_av1_picture_desc *src, + VADecPictureParameterBufferAV1 *dst) +{ + if (!src || !dst) + return false; + + const __typeof__(src->picture_parameter) *p = &src->picture_parameter; + memset(dst, 0, sizeof(*dst)); + + dst->profile = p->profile; + dst->order_hint_bits_minus_1 = p->order_hint_bits_minus_1; + dst->bit_depth_idx = p->bit_depth_idx; + dst->matrix_coefficients = p->matrix_coefficients; + dst->seq_info_fields.fields.still_picture = 0; + dst->seq_info_fields.fields.use_128x128_superblock = + p->seq_info_fields.use_128x128_superblock; + dst->seq_info_fields.fields.enable_filter_intra = + p->seq_info_fields.enable_filter_intra; + dst->seq_info_fields.fields.enable_intra_edge_filter = + p->seq_info_fields.enable_intra_edge_filter; + dst->seq_info_fields.fields.enable_interintra_compound = + p->seq_info_fields.enable_interintra_compound; + dst->seq_info_fields.fields.enable_masked_compound = + p->seq_info_fields.enable_masked_compound; + dst->seq_info_fields.fields.enable_dual_filter = + p->seq_info_fields.enable_dual_filter; + dst->seq_info_fields.fields.enable_order_hint = + p->seq_info_fields.enable_order_hint; + dst->seq_info_fields.fields.enable_jnt_comp = + p->seq_info_fields.enable_jnt_comp; + dst->seq_info_fields.fields.enable_cdef = p->seq_info_fields.enable_cdef; + dst->seq_info_fields.fields.mono_chrome = p->seq_info_fields.mono_chrome; + dst->seq_info_fields.fields.color_range = 0; + dst->seq_info_fields.fields.subsampling_x = + p->seq_info_fields.subsampling_x; + dst->seq_info_fields.fields.subsampling_y = + p->seq_info_fields.subsampling_y; + dst->seq_info_fields.fields.film_grain_params_present = + p->seq_info_fields.film_grain_params_present; + + if (p->frame_width == 0 || p->frame_height == 0) + return false; + dst->frame_width_minus1 = p->frame_width - 1; + dst->frame_height_minus1 = p->frame_height - 1; + memcpy(dst->ref_frame_idx, p->ref_frame_idx, + sizeof(dst->ref_frame_idx)); + dst->primary_ref_frame = p->primary_ref_frame; + dst->order_hint = p->order_hint; + + dst->seg_info.segment_info_fields.bits.enabled = + p->seg_info.segment_info_fields.enabled; + dst->seg_info.segment_info_fields.bits.update_map = + p->seg_info.segment_info_fields.update_map; + dst->seg_info.segment_info_fields.bits.update_data = + p->seg_info.segment_info_fields.update_data; + dst->seg_info.segment_info_fields.bits.temporal_update = + p->seg_info.segment_info_fields.temporal_update; + memcpy(dst->seg_info.feature_data, p->seg_info.feature_data, + sizeof(dst->seg_info.feature_data)); + memcpy(dst->seg_info.feature_mask, p->seg_info.feature_mask, + sizeof(dst->seg_info.feature_mask)); + + dst->tile_cols = p->tile_cols; + dst->tile_rows = p->tile_rows; + if (!dst->tile_cols || !dst->tile_rows || dst->tile_cols > 64 || + dst->tile_rows > 64 || + (uint32_t)dst->tile_cols * dst->tile_rows > 256) + return false; + for (unsigned i = 0; i < dst->tile_cols && i < 63; i++) { + if (!p->width_in_sbs[i]) + return false; + dst->width_in_sbs_minus_1[i] = p->width_in_sbs[i] - 1; + } + for (unsigned i = 0; i < dst->tile_rows && i < 63; i++) { + if (!p->height_in_sbs[i]) + return false; + dst->height_in_sbs_minus_1[i] = p->height_in_sbs[i] - 1; + } + dst->context_update_tile_id = p->context_update_tile_id; + + dst->pic_info_fields.bits.frame_type = p->pic_info_fields.frame_type; + dst->pic_info_fields.bits.show_frame = p->pic_info_fields.show_frame; + dst->pic_info_fields.bits.showable_frame = p->pic_info_fields.showable_frame; + dst->pic_info_fields.bits.error_resilient_mode = + p->pic_info_fields.error_resilient_mode; + dst->pic_info_fields.bits.disable_cdf_update = + p->pic_info_fields.disable_cdf_update; + dst->pic_info_fields.bits.allow_screen_content_tools = + p->pic_info_fields.allow_screen_content_tools; + dst->pic_info_fields.bits.force_integer_mv = + p->pic_info_fields.force_integer_mv; + dst->pic_info_fields.bits.allow_intrabc = p->pic_info_fields.allow_intrabc; + dst->pic_info_fields.bits.use_superres = p->pic_info_fields.use_superres; + dst->pic_info_fields.bits.allow_high_precision_mv = + p->pic_info_fields.allow_high_precision_mv; + dst->pic_info_fields.bits.is_motion_mode_switchable = + p->pic_info_fields.is_motion_mode_switchable; + dst->pic_info_fields.bits.use_ref_frame_mvs = + p->pic_info_fields.use_ref_frame_mvs; + dst->pic_info_fields.bits.disable_frame_end_update_cdf = + p->pic_info_fields.disable_frame_end_update_cdf; + dst->pic_info_fields.bits.uniform_tile_spacing_flag = + p->pic_info_fields.uniform_tile_spacing_flag; + dst->pic_info_fields.bits.allow_warped_motion = + p->pic_info_fields.allow_warped_motion; + dst->pic_info_fields.bits.large_scale_tile = p->pic_info_fields.large_scale_tile; + + dst->superres_scale_denominator = p->superres_scale_denominator; + dst->interp_filter = p->interp_filter; + memcpy(dst->filter_level, p->filter_level, sizeof(dst->filter_level)); + dst->filter_level_u = p->filter_level_u; + dst->filter_level_v = p->filter_level_v; + dst->loop_filter_info_fields.bits.sharpness_level = + p->loop_filter_info_fields.sharpness_level; + dst->loop_filter_info_fields.bits.mode_ref_delta_enabled = + p->loop_filter_info_fields.mode_ref_delta_enabled; + dst->loop_filter_info_fields.bits.mode_ref_delta_update = + p->loop_filter_info_fields.mode_ref_delta_update; + memcpy(dst->ref_deltas, p->ref_deltas, sizeof(dst->ref_deltas)); + memcpy(dst->mode_deltas, p->mode_deltas, sizeof(dst->mode_deltas)); + + dst->base_qindex = p->base_qindex; + dst->y_dc_delta_q = p->y_dc_delta_q; + dst->u_dc_delta_q = p->u_dc_delta_q; + dst->u_ac_delta_q = p->u_ac_delta_q; + dst->v_dc_delta_q = p->v_dc_delta_q; + dst->v_ac_delta_q = p->v_ac_delta_q; + dst->qmatrix_fields.bits.using_qmatrix = p->qmatrix_fields.using_qmatrix; + dst->qmatrix_fields.bits.qm_y = p->qmatrix_fields.qm_y; + dst->qmatrix_fields.bits.qm_u = p->qmatrix_fields.qm_u; + dst->qmatrix_fields.bits.qm_v = p->qmatrix_fields.qm_v; + + dst->mode_control_fields.bits.delta_q_present_flag = + p->mode_control_fields.delta_q_present_flag; + dst->mode_control_fields.bits.log2_delta_q_res = + p->mode_control_fields.log2_delta_q_res; + dst->mode_control_fields.bits.delta_lf_present_flag = + p->mode_control_fields.delta_lf_present_flag; + dst->mode_control_fields.bits.log2_delta_lf_res = + p->mode_control_fields.log2_delta_lf_res; + dst->mode_control_fields.bits.delta_lf_multi = p->mode_control_fields.delta_lf_multi; + dst->mode_control_fields.bits.tx_mode = p->mode_control_fields.tx_mode; + dst->mode_control_fields.bits.reference_select = + p->mode_control_fields.reference_select; + dst->mode_control_fields.bits.reduced_tx_set_used = + p->mode_control_fields.reduced_tx_set_used; + dst->mode_control_fields.bits.skip_mode_present = + p->mode_control_fields.skip_mode_present; + + dst->cdef_damping_minus_3 = p->cdef_damping_minus_3; + dst->cdef_bits = p->cdef_bits; + memcpy(dst->cdef_y_strengths, p->cdef_y_strengths, + sizeof(dst->cdef_y_strengths)); + memcpy(dst->cdef_uv_strengths, p->cdef_uv_strengths, + sizeof(dst->cdef_uv_strengths)); + /* FFmpeg's VA-API backend remaps AV1 restoration values for the VA + * interface: {NONE, SWITCHABLE, WIENER, SGRPROJ}. The bitstream syntax + * uses {NONE, WIENER, SGRPROJ, SWITCHABLE}, so undo that remap here. */ + static const uint8_t restore_remap[4] = { 0, 2, 3, 1 }; + if (p->loop_restoration_fields.yframe_restoration_type > 3 || + p->loop_restoration_fields.cbframe_restoration_type > 3 || + p->loop_restoration_fields.crframe_restoration_type > 3) + return false; + dst->loop_restoration_fields.bits.yframe_restoration_type = + restore_remap[p->loop_restoration_fields.yframe_restoration_type]; + dst->loop_restoration_fields.bits.cbframe_restoration_type = + restore_remap[p->loop_restoration_fields.cbframe_restoration_type]; + dst->loop_restoration_fields.bits.crframe_restoration_type = + restore_remap[p->loop_restoration_fields.crframe_restoration_type]; + dst->loop_restoration_fields.bits.lr_unit_shift = + p->loop_restoration_fields.lr_unit_shift; + dst->loop_restoration_fields.bits.lr_uv_shift = + p->loop_restoration_fields.lr_uv_shift; + + return true; +} + +static void +tva_av1_pending_clear(struct tva_av1_frame *frame) +{ + for (unsigned i = 0; i < frame->picture_count; i++) + free(frame->pictures[i].tile_data); + memset(frame, 0, sizeof(*frame)); +} + +/* The next VA picture exposes the reference-frame map after the pending + * picture. A target appearing in map slot i means that the pending picture + * refreshed slot i; multiple matches are preserved for key-like updates. */ +static uint8_t +tva_av1_refresh_mask(const struct pipe_av1_picture_desc *next, + const struct pipe_video_buffer *target) +{ + if (!next || !target) + return 0; + + uint8_t mask = 0; + for (unsigned i = 0; i < 8; i++) { + if (next->ref[i] == target) + mask |= (uint8_t)(1u << i); + } + return mask; +} + +static int +tva_av1_send_pending(struct tva_codec *c) +{ + struct tva_av1_frame *frame = &c->av1_pending; + if (!frame->valid) + return 0; + + /* Submit one temporal unit per decoded picture, appending a + * show_existing_frame OBU immediately after hidden pictures. Keeping + * the synthetic presentation next to the frame which refreshes the + * reference slot prevents a later temporal unit from changing the map + * before the Qualcomm decoder has emitted the hidden surface. */ + if (tva_av1_inline_show_existing()) { + if (frame->picture_count > UINT32_MAX - c->next_unit) { + debug_printf("tva: AV1 picture sequence overflow\n"); + tva_mark_broken(c); + tva_av1_pending_clear(frame); + return -1; + } + + mtx_lock(&c->pend_mutex); + for (unsigned i = 0; i < frame->picture_count; i++) { + if (frame->pictures[i].pending) + frame->pictures[i].pending->unit_seq = + (uint32_t)(c->next_unit + i + 1); + } + mtx_unlock(&c->pend_mutex); + + for (unsigned i = 0; i < frame->picture_count; i++) { + struct tva_av1_picture *pic = &frame->pictures[i]; + size_t unit_cap = pic->tile_bytes + 8192u + 64u; + if (unit_cap < pic->tile_bytes || unit_cap > MAX_FRAME) { + debug_printf("tva: AV1 picture is too large (%zu bytes)\n", + pic->tile_bytes); + tva_mark_broken(c); + tva_av1_pending_clear(frame); + return -1; + } + + uint8_t *unit = malloc(unit_cap); + if (!unit) { + tva_mark_broken(c); + tva_av1_pending_clear(frame); + return -1; + } + + size_t unit_len = 0; + size_t n = dmd_av1_obu_header(DMD_OBU_TEMPORAL_DELIMITER, 0, + unit, unit_cap); + if (!n) { + free(unit); + tva_mark_broken(c); + tva_av1_pending_clear(frame); + return -1; + } + unit_len += n; + if (i == 0 && frame->include_sequence) { + n = dmd_av1_build_sequence_header(&pic->picture, + unit + unit_len, + unit_cap - unit_len); + if (!n) { + free(unit); + tva_mark_broken(c); + tva_av1_pending_clear(frame); + return -1; + } + unit_len += n; + } + + n = dmd_av1_build_frame(&pic->picture, pic->tiles, + (int)pic->tile_count, + pic->refresh_frame_flags, + unit + unit_len, unit_cap - unit_len); + if (!n) { + free(unit); + tva_mark_broken(c); + tva_av1_pending_clear(frame); + return -1; + } + unit_len += n; + + uint8_t map_idx = 0; + if (!pic->picture.pic_info_fields.bits.show_frame) { + const uint8_t mask = pic->refresh_frame_flags; + map_idx = mask ? (uint8_t)__builtin_ctz(mask) : 0; + n = dmd_av1_build_show_existing(map_idx, + unit + unit_len, + unit_cap - unit_len); + if (!n) { + free(unit); + tva_mark_broken(c); + tva_av1_pending_clear(frame); + return -1; + } + unit_len += n; + } + + TVA_TRACE("AV1 inline picture=%u unit=%u show=%u map=%u refresh=%02x", + i + 1, (unsigned)c->next_unit + 1, + pic->picture.pic_info_fields.bits.show_frame, + map_idx, pic->refresh_frame_flags); + if (tva_session_send_unit(c->sess, unit, unit_len) != TVA_OK) { + debug_printf("tva: AV1 inline send failed: %s\n", + tva_session_last_error(c->sess)); + free(unit); + tva_av1_pending_clear(frame); + tva_mark_broken(c); + return -1; + } + free(unit); + c->next_unit++; + } + + c->av1_sequence_sent |= frame->include_sequence; + tva_av1_pending_clear(frame); + return 0; + } + + /* In hidden-output mode each picture gets its own input unit. The + * Qualcomm decoder otherwise emits only one output for a temporal unit, + * even when all of its frame OBUs carry show_frame=1. Separate units + * preserve the AV1 reference chain while giving the daemon one PTS (and + * therefore one pending surface) per decoded picture. */ + if (tva_av1_output_hidden()) { + if (frame->picture_count > UINT32_MAX - c->next_unit) { + debug_printf("tva: AV1 temporal unit sequence overflow\n"); + tva_mark_broken(c); + tva_av1_pending_clear(frame); + return -1; + } + + for (unsigned i = 0; i < frame->picture_count; i++) { + struct tva_av1_picture *pic = &frame->pictures[i]; + size_t unit_cap = pic->tile_bytes + 8192u + + (size_t)pic->tile_count * 4u; + if (unit_cap < pic->tile_bytes || unit_cap > MAX_FRAME) { + debug_printf("tva: AV1 temporal unit is too large (%zu bytes)\n", + pic->tile_bytes); + tva_mark_broken(c); + tva_av1_pending_clear(frame); + return -1; + } + + uint8_t *unit = malloc(unit_cap); + if (!unit) { + tva_mark_broken(c); + tva_av1_pending_clear(frame); + return -1; + } + + size_t unit_len = 0; + size_t n = dmd_av1_obu_header(DMD_OBU_TEMPORAL_DELIMITER, 0, + unit, unit_cap); + if (!n) { + free(unit); + tva_mark_broken(c); + tva_av1_pending_clear(frame); + return -1; + } + unit_len += n; + + if (i == 0 && frame->include_sequence) { + n = dmd_av1_build_sequence_header(&pic->picture, + unit + unit_len, + unit_cap - unit_len); + if (!n) { + free(unit); + tva_mark_broken(c); + tva_av1_pending_clear(frame); + return -1; + } + unit_len += n; + } + + VADecPictureParameterBufferAV1 picture = pic->picture; + picture.pic_info_fields.bits.show_frame = 1; + n = dmd_av1_build_frame(&picture, pic->tiles, + (int)pic->tile_count, + pic->refresh_frame_flags, + unit + unit_len, unit_cap - unit_len); + if (!n) { + free(unit); + tva_mark_broken(c); + tva_av1_pending_clear(frame); + return -1; + } + unit_len += n; + + TVA_TRACE("AV1 unit len=%zu picture=%u/%u source_show=%u " + "encoded_show=1 refresh=%02x order=%u type=%u%s", + unit_len, i + 1, frame->picture_count, + pic->picture.pic_info_fields.bits.show_frame, + pic->refresh_frame_flags, pic->picture.order_hint, + pic->picture.pic_info_fields.bits.frame_type, + i == 0 && frame->include_sequence ? " sequence" : ""); + if (tva_session_send_unit(c->sess, unit, unit_len) != TVA_OK) { + debug_printf("tva: AV1 send_unit failed: %s\n", + tva_session_last_error(c->sess)); + free(unit); + tva_av1_pending_clear(frame); + tva_mark_broken(c); + return -1; + } + free(unit); + c->next_unit++; + } + + c->av1_sequence_sent |= frame->include_sequence; + tva_av1_pending_clear(frame); + return 0; + } + + size_t tile_bytes = 0; + size_t overhead = 2048; + for (unsigned i = 0; i < frame->picture_count; i++) { + const struct tva_av1_picture *pic = &frame->pictures[i]; + if (pic->tile_bytes > SIZE_MAX - tile_bytes || + tile_bytes + pic->tile_bytes > SIZE_MAX - overhead || + overhead > SIZE_MAX - 1024 - (size_t)pic->tile_count * 4) + goto too_large; + tile_bytes += pic->tile_bytes; + overhead += 1024 + (size_t)pic->tile_count * 4; + } + if (tile_bytes > MAX_FRAME || tile_bytes > SIZE_MAX - overhead || + tile_bytes + overhead > MAX_FRAME) { +too_large: + debug_printf("tva: AV1 temporal unit is too large (%zu bytes)\n", + tile_bytes); + tva_mark_broken(c); + tva_av1_pending_clear(frame); + return -1; + } + + const size_t unit_cap = tile_bytes + overhead; + uint8_t *unit = malloc(unit_cap); + if (!unit) { + tva_mark_broken(c); + tva_av1_pending_clear(frame); + return -1; + } + + size_t unit_len = 0; + size_t n = dmd_av1_obu_header(DMD_OBU_TEMPORAL_DELIMITER, 0, + unit, unit_cap); + if (!n) + goto failed; + unit_len += n; + + if (frame->include_sequence) { + n = dmd_av1_build_sequence_header(&frame->pictures[0].picture, + unit + unit_len, + unit_cap - unit_len); + if (!n) + goto failed; + unit_len += n; + } + + for (unsigned i = 0; i < frame->picture_count; i++) { + struct tva_av1_picture *pic = &frame->pictures[i]; + n = dmd_av1_build_frame(&pic->picture, pic->tiles, + (int)pic->tile_count, + pic->refresh_frame_flags, + unit + unit_len, unit_cap - unit_len); + if (!n) + goto failed; + unit_len += n; + } + + TVA_TRACE("AV1 unit len=%zu pictures=%u payload=%zu%s", unit_len, + frame->picture_count, tile_bytes, + frame->include_sequence ? " sequence" : ""); + for (unsigned i = 0; i < frame->picture_count; i++) + TVA_TRACE("AV1 unit picture=%u show=%u refresh=%02x order=%u type=%u", + i + 1, + frame->pictures[i].picture.pic_info_fields.bits.show_frame, + frame->pictures[i].refresh_frame_flags, + frame->pictures[i].picture.order_hint, + frame->pictures[i].picture.pic_info_fields.bits.frame_type); + + /* A hidden AV1 picture updates the decoder reference map but does not + * normally produce a MediaCodec output buffer. Keep the efficient + * temporal-unit submission above, then ask the decoder to present each + * refreshed reference with a tiny show_existing_frame unit. This fills + * the VA surface assigned to the hidden picture without decoding it a + * second time or forcing every picture into a separate access unit. */ + unsigned synthetic_hidden = 0; + if (tva_av1_synthetic_show_existing()) { + const uint32_t base_seq = (uint32_t)c->next_unit + 1; + mtx_lock(&c->pend_mutex); + for (unsigned i = 0; i < frame->picture_count; i++) { + struct tva_av1_picture *pic = &frame->pictures[i]; + if (!pic->pending) + continue; + if (pic->picture.pic_info_fields.bits.show_frame) { + pic->pending->unit_seq = base_seq; + } else { + pic->pending->unit_seq = base_seq + 1 + synthetic_hidden; + synthetic_hidden++; + } + } + mtx_unlock(&c->pend_mutex); + } + + if (tva_session_send_unit(c->sess, unit, unit_len) != TVA_OK) { + debug_printf("tva: AV1 send_unit failed: %s\n", + tva_session_last_error(c->sess)); + free(unit); + tva_av1_pending_clear(frame); + tva_mark_broken(c); + return -1; + } + + free(unit); + c->av1_sequence_sent |= frame->include_sequence; + c->next_unit++; + + if (tva_av1_synthetic_show_existing()) { + for (unsigned i = 0; i < frame->picture_count; i++) { + struct tva_av1_picture *pic = &frame->pictures[i]; + if (pic->picture.pic_info_fields.bits.show_frame) + continue; + + uint8_t mask = pic->refresh_frame_flags; + uint8_t map_idx = mask ? (uint8_t)__builtin_ctz(mask) : 0; + unsigned char show_unit[32]; + size_t show_hdr = dmd_av1_obu_header(DMD_OBU_TEMPORAL_DELIMITER, + 0, show_unit, + sizeof(show_unit)); + size_t show_obu = show_hdr + ? dmd_av1_build_show_existing(map_idx, + show_unit + show_hdr, + sizeof(show_unit) - show_hdr) + : 0; + if (!show_hdr || !show_obu || + show_hdr + show_obu > sizeof(show_unit)) { + debug_printf("tva: AV1 show_existing_frame build failed\n"); + tva_av1_pending_clear(frame); + tva_mark_broken(c); + return -1; + } + const size_t show_len = show_hdr + show_obu; + TVA_TRACE("AV1 synthetic show_existing unit=%u map=%u refresh=%02x", + (unsigned)c->next_unit + 1, map_idx, mask); + if (tva_session_send_unit(c->sess, show_unit, show_len) != TVA_OK) { + debug_printf("tva: AV1 show_existing send failed: %s\n", + tva_session_last_error(c->sess)); + tva_av1_pending_clear(frame); + tva_mark_broken(c); + return -1; + } + c->next_unit++; + } + } + + tva_av1_pending_clear(frame); + return 0; + +failed: + free(unit); + tva_av1_pending_clear(frame); + tva_mark_broken(c); + return -1; +} + +/* + * Send the accumulated picture to the daemon and register the fence. + * Returns 0 on success, non-zero to make EndPicture report + * VA_STATUS_ERROR_OPERATION_FAILED. + */ +static int +tva_codec_end_frame(struct pipe_video_codec *codec, + struct pipe_video_buffer *target, + struct pipe_picture_desc *picture) +{ + struct tva_codec *c = tva_codec(codec); + + if (tva_codec_is_broken(c)) { + TVA_TRACE("end_frame on broken session"); + return -1; + } + if (!c->sess) + return 0; /* TVA_NO_SESSION dry run */ + + if (!c->acc_len) + return 0; /* an empty VA picture carries no decode work */ + + int codec_id = tva_codec_id(c->base.profile); + enum pipe_video_format format = u_reduce_video_profile(c->base.profile); + struct tva_pending *pending = NULL; + TVA_TRACE("end_frame enter acc=%zu profile=%d format=%d", c->acc_len, + (int)c->base.profile, (int)format); + uint32_t last_vcl = 0; + + if (format == PIPE_VIDEO_FORMAT_MPEG4_AVC || + format == PIPE_VIDEO_FORMAT_HEVC) { + /* ffmpeg does not deliver SPS/PPS as slice data; regenerate them + * from the frontend-parsed structures and send them as CSD units + * whenever they change. The daemon excludes parameter sets from + * the unit index. */ + if (format == PIPE_VIDEO_FORMAT_MPEG4_AVC) { + struct pipe_h264_picture_desc *h264 = + (struct pipe_h264_picture_desc *)picture; + if (h264 && h264->pps && h264->pps->sps) { + uint8_t *sps_rbsp = NULL, *pps_rbsp = NULL; + static const uint8_t sc[4] = { 0, 0, 0, 1 }; + if (!c->h264_pps_defaults_valid) { + /* VA-API does not preserve the slice override flag, so + * the per-picture list lengths cannot be used as PPS + * defaults. The DPB size is also not the same thing as + * the PPS active-list defaults: Bilibili and Chromium + * streams commonly use two L0 references and no L1 + * references while advertising a larger DPB. Keep a + * conservative, stable default instead of emitting a + * six-reference PPS that Qualcomm MediaCodec rejects + * when a slice relies on the original PPS default. */ + unsigned default_l0 = c->base.max_references + ? MIN2(c->base.max_references, 3) - 1 + : 0; + c->h264_pps_l0_default = default_l0; + c->h264_pps_l1_default = 0; + c->h264_pps_defaults_valid = true; + } + size_t sps_rbsp_len = tva_build_h264_sps(c->base.profile, + h264->pps->sps, + c->base.max_references, + target ? target->width : c->base.width, + target ? target->height : c->base.height, + &sps_rbsp); + size_t pps_rbsp_len = tva_build_h264_pps(c->base.profile, + h264->pps, + c->h264_pps_l0_default, + c->h264_pps_l1_default, + &pps_rbsp); + if (!sps_rbsp_len || !pps_rbsp_len) { + debug_printf("tva: CSD synthesis failed\n"); + free(sps_rbsp); + free(pps_rbsp); + tva_mark_broken(c); + return -1; + } + size_t sps_len = sps_rbsp_len + 4; + size_t pps_len = pps_rbsp_len + 4; + uint8_t *sps_buf = malloc(sps_len); + uint8_t *pps_buf = malloc(pps_len); + if (!sps_buf || !pps_buf) { + free(sps_buf); free(pps_buf); free(sps_rbsp); free(pps_rbsp); + tva_mark_broken(c); + return -1; + } + memcpy(sps_buf, sc, 4); + memcpy(sps_buf + 4, sps_rbsp, sps_rbsp_len); + memcpy(pps_buf, sc, 4); + memcpy(pps_buf + 4, pps_rbsp, pps_rbsp_len); + free(sps_rbsp); + free(pps_rbsp); + + size_t nlen = sps_len + pps_len; + uint8_t *csd = malloc(nlen); + if (!csd) { + free(sps_buf); + free(pps_buf); + tva_mark_broken(c); + return -1; + } + memcpy(csd, sps_buf, sps_len); + memcpy(csd + sps_len, pps_buf, pps_len); + + if (!c->csd || c->csd_len != nlen || + memcmp(c->csd, csd, nlen)) { + free(c->csd); + c->csd = csd; + c->csd_len = nlen; + csd = NULL; + int rc1 = tva_session_send_unit(c->sess, sps_buf, sps_len); + int rc2 = rc1 == TVA_OK + ? tva_session_send_unit(c->sess, pps_buf, pps_len) + : rc1; + if (rc2 != TVA_OK) { + debug_printf("tva: CSD send failed: %s\n", + tva_session_last_error(c->sess)); + tva_mark_broken(c); + } else { + TVA_TRACE("CSD sent: profile=%d sps=%zu pps=%zu maxrefs=%u defaults=%u/%u", + (int)c->base.profile, sps_len, pps_len, + c->base.max_references, + c->h264_pps_l0_default, + c->h264_pps_l1_default); + } + } + free(csd); + free(sps_buf); + free(pps_buf); + } + if (tva_codec_is_broken(c)) + return -1; + } else if (format == PIPE_VIDEO_FORMAT_HEVC) { + struct pipe_h265_picture_desc *h265 = + (struct pipe_h265_picture_desc *)picture; + if (h265 && h265->pps && h265->pps->sps) { + const struct pipe_h265_sps *sps = h265->pps->sps; + static const uint8_t sc[4] = { 0, 0, 0, 1 }; + uint8_t *vps_nalu = NULL, *sps_nalu = NULL; + uint8_t *pps_nalu = NULL; + + if (!tva_h265_can_build(sps, h265)) { + debug_printf("tva: HEVC parameter sets cannot be synthesized " + "for this stream\n"); + tva_mark_broken(c); + return -1; + } + + size_t vps_nalu_len = + tva_build_h265_vps(sps, c->base.profile, &vps_nalu); + size_t sps_nalu_len = + tva_build_h265_sps(sps, c->base.profile, &sps_nalu); + size_t pps_nalu_len = tva_build_h265_pps(h265->pps, &pps_nalu); + if (!vps_nalu_len || !sps_nalu_len || !pps_nalu_len) { + debug_printf("tva: HEVC CSD synthesis failed\n"); + free(vps_nalu); + free(sps_nalu); + free(pps_nalu); + tva_mark_broken(c); + return -1; + } + + size_t vps_len = vps_nalu_len + 4; + size_t sps_len = sps_nalu_len + 4; + size_t pps_len = pps_nalu_len + 4; + size_t nlen = vps_len + sps_len + pps_len; + uint8_t *csd = malloc(nlen); + if (!csd) { + free(vps_nalu); + free(sps_nalu); + free(pps_nalu); + tva_mark_broken(c); + return -1; + } + memcpy(csd, sc, 4); + memcpy(csd + 4, vps_nalu, vps_nalu_len); + memcpy(csd + vps_len, sc, 4); + memcpy(csd + vps_len + 4, sps_nalu, sps_nalu_len); + memcpy(csd + vps_len + sps_len, sc, 4); + memcpy(csd + vps_len + sps_len + 4, + pps_nalu, pps_nalu_len); + + if (!c->csd || c->csd_len != nlen || + memcmp(c->csd, csd, nlen)) { + free(c->csd); + c->csd = csd; + c->csd_len = nlen; + csd = NULL; + + int rc = tva_session_send_unit(c->sess, + c->csd, vps_len); + if (rc == TVA_OK) + rc = tva_session_send_unit(c->sess, + c->csd + vps_len, sps_len); + if (rc == TVA_OK) + rc = tva_session_send_unit(c->sess, + c->csd + vps_len + sps_len, + pps_len); + if (rc != TVA_OK) { + debug_printf("tva: HEVC CSD send failed: %s\n", + tva_session_last_error(c->sess)); + tva_mark_broken(c); + } else { + TVA_TRACE("HEVC CSD sent: profile=%d vps=%zu sps=%zu pps=%zu", + (int)c->base.profile, vps_len, sps_len, + pps_len); + } + } + free(csd); + free(vps_nalu); + free(sps_nalu); + free(pps_nalu); + } + if (tva_codec_is_broken(c)) + return -1; + } + + /* A VA picture may contain several slice NALUs. MediaCodec consumes + * an access unit, not an individual slice: sending each slice as a + * separate input buffer gives the decoder several different PTS + * values for one output frame, and the value returned on that frame + * is implementation dependent (usually the first or last slice). + * Keep the NALUs together so one daemon unit and one PTS identify the + * whole picture. */ + bool have_vcl = false; + size_t scan = 0; + while (scan < c->acc_len) { + size_t sc = tva_next_start_code(c->acc, c->acc_len, scan); + if (sc >= c->acc_len) + break; + size_t next = tva_next_start_code(c->acc, c->acc_len, sc + 3); + if (next > c->acc_len) + next = c->acc_len; + size_t end = next; + if (end < c->acc_len) { + while (end > sc + 3 && c->acc[end - 1] == 0) + end--; + } + if (!tva_is_param_set(codec_id, c->acc + sc, end - sc)) + have_vcl = true; + scan = next; + } + if (have_vcl && c->next_unit >= UINT32_MAX) { + debug_printf("tva: input unit sequence exhausted\n"); + c->acc_len = 0; + tva_mark_broken(c); + return -1; + } + uint32_t picture_unit = have_vcl ? (uint32_t)c->next_unit + 1 + : (uint32_t)c->next_unit; + if (c->acc_len > MAX_FRAME) { + debug_printf("tva: picture access unit too large (%zu bytes)\n", + c->acc_len); + c->acc_len = 0; + tva_mark_broken(c); + return -1; + } + if (have_vcl) { + if (picture_unit == 0) { + tva_mark_broken(c); + return -1; + } + struct tva_fence *fence = CALLOC_STRUCT(tva_fence); + if (!fence) { + tva_mark_broken(c); + return -1; + } + mtx_lock(&c->pend_mutex); + pending = tva_pend_reserve_locked(c, picture_unit, target, fence); + mtx_unlock(&c->pend_mutex); + if (!pending) { + FREE(fence); + c->acc_len = 0; + return -1; + } + if (picture && picture->out_fence) { + if (*picture->out_fence) + tva_codec_destroy_fence(codec, *picture->out_fence); + *picture->out_fence = (struct pipe_fence_handle *)fence; + } + } + + TVA_TRACE("access unit len=%zu vcl=%d unit=%u", c->acc_len, + have_vcl, picture_unit); + int r = tva_session_send_unit(c->sess, c->acc, c->acc_len); + if (r != TVA_OK) { + debug_printf("tva: send_unit failed: %s\n", + tva_session_last_error(c->sess)); + tva_mark_broken(c); + return -1; + } + if (have_vcl) { + c->next_unit++; + last_vcl = (uint32_t)c->next_unit; + } + } else if (format == PIPE_VIDEO_FORMAT_AV1) { + /* VA-API supplies AV1 tile payloads separately from the structured + * picture parameters. Rebuild complete temporal units before + * passing them to MediaCodec; forwarding c->acc directly is not valid + * AV1 and starts with tile bytes rather than an OBU header. */ + const struct pipe_av1_picture_desc *av1 = + (const struct pipe_av1_picture_desc *)picture; + VADecPictureParameterBufferAV1 va_pic; + if (!tva_av1_to_va_picture(av1, &va_pic)) { + debug_printf("tva: AV1 picture parameters are invalid\n"); + c->acc_len = 0; + tva_mark_broken(c); + return -1; + } + const bool show_frame = va_pic.pic_info_fields.bits.show_frame; + TVA_TRACE("AV1 VA params: warp=%u tx=%u show=%u refresh=%u", + va_pic.pic_info_fields.bits.allow_warped_motion, + va_pic.mode_control_fields.bits.reduced_tx_set_used, + show_frame, + av1->picture_parameter.refresh_frame_flags); + + const uint32_t tile_count = + (uint32_t)va_pic.tile_cols * (uint32_t)va_pic.tile_rows; + if (!tile_count || tile_count > 256 || + av1->slice_parameter.slice_count != tile_count) { + debug_printf("tva: AV1 tile count mismatch (params=%u slices=%u)\n", + tile_count, + av1 ? (unsigned)av1->slice_parameter.slice_count : 0); + c->acc_len = 0; + tva_mark_broken(c); + return -1; + } + + struct dmd_av1_tile tiles[256]; + size_t tile_bytes = 0; + for (uint32_t i = 0; i < tile_count; i++) { + const size_t off = av1->slice_parameter.slice_data_offset[i]; + const size_t len = av1->slice_parameter.slice_data_size[i]; + if (!len || off > c->acc_len || len > c->acc_len - off) { + debug_printf("tva: AV1 tile %u is outside the slice buffer " + "(off=%zu len=%zu buffer=%zu)\n", + i, off, len, c->acc_len); + c->acc_len = 0; + tva_mark_broken(c); + return -1; + } + if (tile_bytes > SIZE_MAX - len) { + c->acc_len = 0; + tva_mark_broken(c); + return -1; + } + tiles[i].data = c->acc + off; + tiles[i].len = len; + tile_bytes += len; + } + + struct tva_av1_frame *frame = &c->av1_pending; + + /* The VA descriptor has no temporal-delimiter marker. Chromium's + * AV1 parser presents each temporal unit as either one displayed + * frame, or a run of hidden reference frames followed by one displayed + * frame. Once a displayed frame has been seen without a hidden frame + * after it, the next descriptor (hidden or displayed) starts the next + * temporal unit. This keeps a displayed key frame separate from the + * hidden references of the following unit. */ + bool boundary = frame->valid && frame->picture_count && + frame->have_show_frame && + !frame->hidden_since_show_frame; + if (frame->valid && frame->picture_count) { + struct tva_av1_picture *previous = + &frame->pictures[frame->picture_count - 1]; + previous->refresh_frame_flags = tva_av1_refresh_mask(av1, + previous->target); + } + if (boundary) { + TVA_TRACE("AV1 group flush before show frame: pictures=%u", + frame->picture_count); + if (tva_av1_send_pending(c)) { + c->acc_len = 0; + return -1; + } + frame = &c->av1_pending; + } + if (frame->picture_count >= TVA_AV1_GROUP_MAX_FRAMES) { + debug_printf("tva: AV1 temporal unit has too many frames (%u)\n", + frame->picture_count); + c->acc_len = 0; + tva_mark_broken(c); + return -1; + } + + const bool key_frame = va_pic.pic_info_fields.bits.frame_type == 0; + const bool include_sequence = key_frame || !c->av1_sequence_sent; + uint8_t *tile_data = malloc(tile_bytes); + if (!tile_data) { + c->acc_len = 0; + tva_mark_broken(c); + return -1; + } + size_t tile_offset = 0; + for (uint32_t i = 0; i < tile_count; i++) { + memcpy(tile_data + tile_offset, tiles[i].data, tiles[i].len); + tiles[i].data = tile_data + tile_offset; + tile_offset += tiles[i].len; + } + + struct tva_av1_picture *current = + &frame->pictures[frame->picture_count]; + memset(current, 0, sizeof(*current)); + current->target = target; + current->picture = va_pic; + current->tile_data = tile_data; + current->tile_bytes = tile_bytes; + current->tile_count = tile_count; + for (uint32_t i = 0; i < tile_count; i++) + current->tiles[i] = tiles[i]; + + frame->valid = true; + frame->picture_count++; + frame->include_sequence |= include_sequence; + if (show_frame) { + frame->have_show_frame = true; + frame->hidden_since_show_frame = false; + } else { + frame->hidden_since_show_frame = true; + } + + /* Hidden AV1 frames normally update the decoder's reference state but + * do not produce an output buffer. In hidden-output mode the + * reconstructed stream marks them as shown, so reserve a fence-less + * entry for each one and let the reader pair repeated output frames + * with the entries sharing this temporal-unit sequence number. */ + struct tva_fence *fence = NULL; + if (show_frame || tva_av1_output_hidden() || + tva_av1_synthetic_show_existing() || + tva_av1_inline_show_existing()) { + fence = CALLOC_STRUCT(tva_fence); + if (!fence) { + tva_av1_pending_clear(frame); + c->acc_len = 0; + tva_mark_broken(c); + return -1; + } + if (c->next_unit >= UINT32_MAX) { + FREE(fence); + tva_av1_pending_clear(frame); + c->acc_len = 0; + tva_mark_broken(c); + return -1; + } + /* In hidden-output mode pictures are submitted as separate + * daemon units. Reserve the sequence number this picture will + * receive within the accumulated temporal unit; the normal path + * keeps one sequence number for the whole group. */ + uint64_t pending_seq = c->next_unit + 1; + if (tva_av1_output_hidden()) + pending_seq = c->next_unit + frame->picture_count; + else if (tva_av1_synthetic_show_existing() || + tva_av1_inline_show_existing()) + pending_seq = 0; /* assigned when the group is flushed */ + if (pending_seq > UINT32_MAX) { + FREE(fence); + tva_av1_pending_clear(frame); + c->acc_len = 0; + tva_mark_broken(c); + return -1; + } + mtx_lock(&c->pend_mutex); + pending = tva_pend_reserve_locked(c, (uint32_t)pending_seq, + target, fence); + /* Chromium recycles AV1 surfaces as soon as their VA fence is + * replaced. A late daemon output must never be copied into that + * surface, including the ordinary visible-frame path: doing so + * can overwrite a newer frame and make old images flash back on + * screen. */ + if (pending) + pending->drop_on_fence_destroy = true; + mtx_unlock(&c->pend_mutex); + if (!pending) { + FREE(fence); + tva_av1_pending_clear(frame); + c->acc_len = 0; + return -1; + } + current->pending = pending; + } + + TVA_TRACE("AV1 group append picture=%u show=%u tiles=%u payload=%zu%s", + frame->picture_count, show_frame, tile_count, tile_bytes, + show_frame ? " output" : " hidden"); + + if (picture && picture->out_fence && + (show_frame || tva_av1_output_hidden() || + tva_av1_synthetic_show_existing() || + tva_av1_inline_show_existing())) { + if (*picture->out_fence) + tva_codec_destroy_fence(codec, *picture->out_fence); + *picture->out_fence = (struct pipe_fence_handle *)fence; + } else if (picture && picture->out_fence) { + /* A hidden frame has no corresponding output in the normal mode. + * Keep the historical fence-less behaviour there; in + * hidden-output mode it receives the regular fence above so an + * export of a later show_existing_frame surface waits for the + * reference copy. */ + if (*picture->out_fence) + tva_codec_destroy_fence(codec, *picture->out_fence); + *picture->out_fence = NULL; + } + if (show_frame || tva_av1_output_hidden() || + tva_av1_synthetic_show_existing() || + tva_av1_inline_show_existing()) + last_vcl = pending && pending->unit_seq ? pending->unit_seq : + (uint32_t)(c->next_unit + 1); + } else { + /* Chromium splits VP9 superframes into individual pictures. Hidden + * reference pictures must reach MediaCodec, but they do not produce + * an output buffer. Reserving a pending entry for them would leave a + * permanent hole in the queue and eventually exhaust it. */ + const struct pipe_vp9_picture_desc *vp9 = + (const struct pipe_vp9_picture_desc *)picture; + const bool show_frame = !vp9 || + vp9->picture_parameter.pic_fields.show_frame; + if (c->next_unit >= UINT32_MAX) { + debug_printf("tva: VP9 input unit sequence exhausted\n"); + c->acc_len = 0; + tva_mark_broken(c); + return -1; + } + const uint32_t unit_seq = (uint32_t)c->next_unit + 1; + struct tva_fence *fence = NULL; + if (show_frame) { + fence = CALLOC_STRUCT(tva_fence); + if (!fence) + return -1; + mtx_lock(&c->pend_mutex); + pending = tva_pend_reserve_locked(c, unit_seq, target, fence); + mtx_unlock(&c->pend_mutex); + if (!pending) { + FREE(fence); + c->acc_len = 0; + return -1; + } + } + if (picture && picture->out_fence) { + if (*picture->out_fence) + tva_codec_destroy_fence(codec, *picture->out_fence); + *picture->out_fence = show_frame + ? (struct pipe_fence_handle *)fence : NULL; + } + TVA_TRACE("sending whole VP9 frame, len=%zu unit=%u show=%d", + c->acc_len, unit_seq, show_frame); + int r = tva_session_send_unit(c->sess, c->acc, c->acc_len); + if (r != TVA_OK) { + debug_printf("tva: send_unit failed: %s\n", + tva_session_last_error(c->sess)); + mtx_lock(&c->pend_mutex); + tva_mark_broken_locked(c); + mtx_unlock(&c->pend_mutex); + return -1; + } + c->next_unit++; + if (show_frame) + last_vcl = unit_seq; + } + + if (!last_vcl) { + /* picture contained only parameter sets: nothing to wait for */ + c->acc_len = 0; + return 0; + } + + c->acc_len = 0; + unsigned pending_count; + mtx_lock(&c->pend_mutex); + pending_count = c->pend_count; + mtx_unlock(&c->pend_mutex); + TVA_TRACE("end_frame ok: pic unit=%u pending=%u", last_vcl, pending_count); + return 0; +} + +static void +tva_codec_flush(struct pipe_video_codec *codec) +{ + struct tva_codec *c = tva_codec(codec); + if (!c->av1_pending.valid || tva_codec_is_broken(c)) + return; + + /* No following VA descriptor is available at end of stream. The final + * picture does not need to be referenced by a later picture, so its zero + * mask is sufficient; key frames infer all slots and do not code it. */ + if (c->av1_pending.picture_count) + c->av1_pending.pictures[c->av1_pending.picture_count - 1] + .refresh_frame_flags = 0; + (void)tva_av1_send_pending(c); +} + +/* + * Wait for the frame behind `fence` and write it into the surface. Returns + * 1 when the fence is (or became) signaled, 0 on timeout. timeout is in + * nanoseconds, as passed by the VA frontend's _vlVaSyncSurface. + */ +static int +tva_codec_fence_wait(struct pipe_video_codec *codec, + struct pipe_fence_handle *fence_handle, + uint64_t timeout) +{ + struct tva_codec *c = tva_codec(codec); + struct tva_fence *fence = (struct tva_fence *)fence_handle; + + if (!fence) + return 1; + + struct tva_pending *p; + + /* timeout comes in ns from the frontend; VA_TIMEOUT_INFINITE arrives as + * (uint64_t)-1 and MUST be treated as unbounded - casting it to int64 + * yields -1, which would make the deadline expire immediately and fail + * every vaSyncSurface. The reader thread stages frames; this thread + * only waits on the condition variable and copies the staged frame into + * the surface (pipe_context stays on the application thread). */ + bool infinite = timeout == UINT64_MAX; + uint64_t deadline_ns = infinite ? UINT64_MAX : + os_time_get_nano() + timeout; + int ret = 1; + + mtx_lock(&c->pend_mutex); + if (fence->failed) { + mtx_unlock(&c->pend_mutex); + return 0; + } + p = fence->slot; + if (!p) { + mtx_unlock(&c->pend_mutex); + return 1; + } + p->waiters++; + while (!p->ready) { + if (!infinite && timeout == 0) { + ret = 0; + break; + } + uint64_t now = os_time_get_nano(); + if (!infinite && now >= deadline_ns) { + ret = 0; + break; + } + uint64_t wake_ns = infinite ? now + 200000000ull : + MIN2(deadline_ns, now + 200000000ull); + struct timespec ts; + timespec_from_nsec(&ts, wake_ns); + u_cnd_monotonic_timedwait(&c->pend_cond, &c->pend_mutex, &ts); + } + + TVA_TRACE("fence unit=%u ready=%d failed=%d copied=%d staging=%p", + p->unit_seq, p->ready, p->failed, p->copied, (void *)p->staging); + + if (p->ready && !p->failed && !p->copied && p->staging) { + p->copied = tva_copy_frame(c, p); + TVA_TRACE("fence copy unit=%u result=%d", p->unit_seq, p->copied); + if (!p->copied) + p->failed = true; + } + tva_notify_frame_resources(c, p); + if (p->failed) + ret = 0; + p->waiters--; + u_cnd_monotonic_broadcast(&c->pend_cond); + mtx_unlock(&c->pend_mutex); + return ret; +} + +static void +tva_codec_destroy_fence(struct pipe_video_codec *codec, + struct pipe_fence_handle *fence_handle) +{ + struct tva_codec *c = tva_codec(codec); + struct tva_fence *fence = (struct tva_fence *)fence_handle; + + if (!fence) + return; + mtx_lock(&c->pend_mutex); + /* Chromium destroys a surface fence immediately before reusing the VA + * surface for another picture. Do not let a reordered daemon output + * arrive after that reuse: it would overwrite the newer picture and make + * an already displayed frame flash back on screen. Wait for the pending + * output while the resource is still owned by this entry, then copy a + * staged frame before detaching the client fence. */ + bool fail = false; + struct tva_pending *p = fence->slot; + if (p && p->drop_on_fence_destroy) { + const unsigned wait_ms = tva_av1_fence_destroy_wait_ms(); + const uint64_t wait_start = os_time_get_nano(); + const uint64_t deadline = os_time_get_nano() + + (uint64_t)wait_ms * 1000000ull; + /* Keep the entry alive while waiting, but release pend_mutex so the + * reader can publish the matching frame. Waiting on pend_cond while + * holding this mutex would deadlock: the reader needs the same mutex + * to set p->ready and signal the condition. */ + p->waiters++; + mtx_unlock(&c->pend_mutex); + for (;;) { + mtx_lock(&c->pend_mutex); + bool done = p->ready || c->broken; + mtx_unlock(&c->pend_mutex); + uint64_t now = os_time_get_nano(); + if (done || now >= deadline) + break; + uint64_t remaining_us = (deadline - now) / 1000ull; + os_time_sleep((int64_t)MIN2(remaining_us, 1000ull)); + } + mtx_lock(&c->pend_mutex); + const uint64_t wait_elapsed = os_time_get_nano() - wait_start; + if (!p->ready || wait_elapsed >= 1000000ull) + TVA_TRACE("recycled AV1 fence wait unit=%u ready=%d wait_ms=%u " + "elapsed=%.3f ms", p->unit_seq, p->ready, wait_ms, + (double)wait_elapsed / 1000000.0); + if (!p->ready) { + TVA_TRACE("recycled AV1 fence timed out unit=%u", p->unit_seq); + fail = true; + } else if (!p->failed && !p->copied && p->staging) { + p->copied = tva_copy_frame(c, p); + TVA_TRACE("recycled AV1 fence copy unit=%u result=%d", + p->unit_seq, p->copied); + if (!p->copied) + fail = true; + } + p->waiters--; + } + tva_detach_fence_locked(fence, fail); + u_cnd_monotonic_broadcast(&c->pend_cond); + mtx_unlock(&c->pend_mutex); + FREE(fence); +} + +static void +tva_codec_destroy(struct pipe_video_codec *codec) +{ + struct tva_codec *c = tva_codec(codec); + + mtx_lock(&c->pend_mutex); + c->quitting = true; + c->broken = true; + tva_fail_pending_locked(c); + u_cnd_monotonic_broadcast(&c->pend_cond); + mtx_unlock(&c->pend_mutex); + tva_session_cancel(c->sess); + if (c->reader_started) + thrd_join(c->reader, NULL); + mtx_lock(&c->pend_mutex); + while (c->pend_count) + tva_pend_pop_locked(c); + mtx_unlock(&c->pend_mutex); + tva_av1_pending_clear(&c->av1_pending); + mtx_destroy(&c->pend_mutex); + u_cnd_monotonic_destroy(&c->pend_cond); + TVA_TRACE("codec destroy: %llu units, %llu frames", (unsigned long long)c->next_unit, (unsigned long long)c->frames_done); + tva_session_destroy(c->sess); + free(c->csd); + free(c->acc); + FREE(c); +} + +static struct pipe_video_buffer * +tva_pipe_create_video_buffer(struct pipe_context *context, + const struct pipe_video_buffer *templat) +{ + fprintf(stderr, "tva: create video buffer format=%d size=%ux%u bind=%#x\n", + templat ? templat->buffer_format : -1, + templat ? templat->width : 0, templat ? templat->height : 0, + templat ? templat->bind : 0); +#if defined(__linux__) + if (tva_contiguous_dmabuf_enabled()) { + struct pipe_video_buffer *buffer = + tva_create_contiguous_video_buffer(context, templat); + if (buffer) + { + fprintf(stderr, "tva: contiguous video buffer ready\n"); + return buffer; + } + fprintf(stderr, "tva: contiguous video buffer failed\n"); + if (getenv("DMD_VA_LOG")) + fprintf(stderr, "tva: contiguous NV12 allocation failed; " + "falling back to separate plane resources\n"); + } +#endif + + /* Bridge surfaces are CPU-filled and may be exported to Vulkan. Allocate + * them as linear shared resources so KGSL does not need an export-time + * shadow allocation. */ + struct pipe_video_buffer bridge_templ = *templat; + bridge_templ.bind |= PIPE_BIND_SHARED | PIPE_BIND_LINEAR; + struct pipe_video_buffer *buffer = + vl_video_buffer_create(context, &bridge_templ); + fprintf(stderr, "tva: separate video buffer %s\n", buffer ? "ready" : "failed"); + return buffer; +} + +static struct pipe_video_buffer * +tva_pipe_create_video_buffer_with_modifiers( + struct pipe_context *context, const struct pipe_video_buffer *templat, + const uint64_t *modifiers, unsigned modifiers_count) +{ + /* Explicit modifiers cannot change the bridge's linear CPU-copy layout. */ + (void)modifiers; + (void)modifiers_count; + return tva_pipe_create_video_buffer(context, templat); +} + +static struct pipe_video_codec * +tva_pipe_create_video_codec(struct pipe_context *context, + const struct pipe_video_codec *templat) +{ + fprintf(stderr, "tva: create codec profile=%d entrypoint=%d size=%ux%u\n", + templat ? templat->profile : -1, + templat ? templat->entrypoint : -1, + templat ? templat->width : 0, templat ? templat->height : 0); + if (templat->entrypoint == PIPE_VIDEO_ENTRYPOINT_PROCESSING) + return vl_compositor_create_proc(context, false); + + if (templat->entrypoint != PIPE_VIDEO_ENTRYPOINT_BITSTREAM || + !tva_profile_supported(templat->profile)) + { + fprintf(stderr, "tva: reject codec profile=%d entrypoint=%d\n", + templat->profile, templat->entrypoint); + return NULL; /* no encode / unsupported profiles through the bridge */ + } + + int codec_id = tva_codec_id(templat->profile); + if (codec_id < 0) + { + fprintf(stderr, "tva: reject codec id for profile=%d\n", templat->profile); + return NULL; + } + + unsigned pipeline_depth = tva_pipeline_depth_default; + const char *d = getenv("TERMUX_VA_PIPELINE_DEPTH"); + if (d && *d) { + long v = atol(d); + if (v >= 2 && v <= DMD_PIPELINE_DEPTH_MAX) + pipeline_depth = (unsigned)v; + } else if (codec_id == CODEC_AV1 && tva_av1_output_hidden()) { + /* Hidden-output AV1 produces one daemon result for every reference + * and displayed frame. Use the complete SHM pool by default so the + * sixth pending surface does not add avoidable backpressure. */ + pipeline_depth = SHM_SLOTS; + } + + struct tva_codec *c = CALLOC_STRUCT(tva_codec); + if (!c) + { + fprintf(stderr, "tva: codec allocation failed\n"); + return NULL; + } + + struct tva_session_config cfg; + tva_session_config_defaults(&cfg); + cfg.codec = codec_id; + cfg.width = templat->width; + cfg.height = templat->height; + /* SHM zero-copy on by default; DMD_WANT_SHM=0 disables (legacy name + * kept on purpose) */ + const char *shm = getenv("DMD_WANT_SHM"); + cfg.want_shm = !(shm && !strcmp(shm, "0")); + if (cfg.want_shm && pipeline_depth > SHM_SLOTS) + pipeline_depth = SHM_SLOTS; + + struct tva_error err; + memset(&err, 0, sizeof(err)); + + tva_dbg_seq++; + TVA_TRACE("session create codec=%d %dx%d", codec_id, templat->width, templat->height); + c->sess = tva_session_create(&cfg, &err); + if (!c->sess) { + debug_printf("tva: session create failed: %s\n", + err.msg[0] ? err.msg : "unknown error (set DMD_VA_LOG=1)"); + FREE(c); + return NULL; + } + + c->pipe = context; + c->next_unit = 0; + c->pipeline_depth = pipeline_depth; + /* DMD_AV1_STRICT_PENDING=1 keeps AV1 output submissions within the + * configured depth. The default soft limit lets the pending cache absorb + * short decoder bursts while fence recycling still protects old surfaces. */ + c->strict_pending = codec_id == CODEC_AV1 && tva_av1_strict_pending(); + + mtx_init(&c->pend_mutex, mtx_plain); + if (u_cnd_monotonic_init(&c->pend_cond) != thrd_success) { + tva_session_destroy(c->sess); + free(c->csd); + free(c->acc); + mtx_destroy(&c->pend_mutex); + FREE(c); + return NULL; + } + if (thrd_create(&c->reader, tva_reader_thread, c) == thrd_success) + c->reader_started = true; + + c->base.context = context; + c->base.profile = templat->profile; + c->base.level = templat->level; + c->base.entrypoint = templat->entrypoint; + c->base.width = templat->width; + c->base.height = templat->height; + c->base.max_references = templat->max_references; + c->base.destroy = tva_codec_destroy; + c->base.begin_frame = tva_codec_begin_frame; + c->base.decode_macroblock = NULL; + c->base.decode_bitstream = tva_codec_decode_bitstream; + c->base.end_frame = tva_codec_end_frame; + c->base.flush = tva_codec_flush; + c->base.get_feedback = NULL; + c->base.fence_wait = tva_codec_fence_wait; + c->base.destroy_fence = tva_codec_destroy_fence; + + TVA_TRACE("codec ready pipeline=%u strict=%d hidden=%d synthetic=%d inline=%d shm=%d", + c->pipeline_depth, c->strict_pending, + tva_av1_output_hidden(), tva_av1_synthetic_show_existing(), + tva_av1_inline_show_existing(), cfg.want_shm); + return &c->base; +} + +/* ------------------------------------------------------ pipe codec hooks */ +/* + * The multimedia context's create_video_codec / create_video_buffer hooks + * are filled in directly on the real context: drivers without a video path + * leave them NULL, and the bridge implementations above receive the real + * context pointer, so no wrapper object is involved anywhere. + */ +void +tva_bridge_pipe_set_codec_hooks(struct pipe_context *pipe) +{ + if (!pipe) + return; + pipe->create_video_codec = tva_pipe_create_video_codec; + pipe->create_video_buffer = tva_pipe_create_video_buffer; + pipe->create_video_buffer_with_modifiers = + tva_pipe_create_video_buffer_with_modifiers; +} + +/* ------------------------------------------------- vscreen creation */ +/* + * The stock loader path breaks on the kgsl stack in two ways: + * + * 1. Xiaomi/DroidSpaces kernels report the display controller's DRM node + * as "msm_drm" - no pipe_loader descriptor matches that name, so the + * kmsro/zink fallbacks engage and fail (the kgsl stack has no usable + * Vulkan device for zink, hence "ZINK: failed to choose pdev"). + * 2. Even on kernels reporting "msm", the msm kmd drives no GPU there - + * the Adreno GPU is only reachable through /dev/kgsl-3d0. + * + * The fork registers a "kgsl" alias of the freedreno descriptor whose + * device layer redirects GPU submission to /dev/kgsl-3d0 while keeping the + * handed fd as the control/identity fd (freedreno_device.c); EGL uses + * exactly that override via MESA_LOADER_DRIVER_OVERRIDE=kgsl. The bridge + * does the same for VA-API by re-pointing the probed device's driver_name. + */ + +static void +tva_vscreen_destroy(struct vl_screen *vscreen) +{ + vscreen->pscreen->destroy(vscreen->pscreen); + pipe_loader_release(&vscreen->dev, 1); + FREE(vscreen); +} + +static struct vl_screen * +tva_vscreen_from_pscreen(struct pipe_screen *pscreen, + struct pipe_loader_device *dev) +{ + struct vl_screen *vscreen = CALLOC_STRUCT(vl_screen); + if (!vscreen) { + pscreen->destroy(pscreen); + pipe_loader_release(&dev, 1); + return NULL; + } + + vscreen->pscreen = pscreen; + vscreen->dev = dev; + vscreen->destroy = tva_vscreen_destroy; + vscreen->texture_from_drawable = NULL; + vscreen->get_dirty_area = NULL; + vscreen->get_timestamp = NULL; + vscreen->set_next_timestamp = NULL; + vscreen->get_private = NULL; + vscreen->set_back_texture_from_output = NULL; + return vscreen; +} + +/* llvmpipe over the null sw winsys: no GPU needed, enough for the CPU + * frame-copy paths (vainfo, ffmpeg vaMapBuffer). */ +static struct vl_screen * +tva_vscreen_sw(void) +{ + struct pipe_loader_device *dev = NULL; + if (!pipe_loader_sw_probe_null(&dev)) + return NULL; + + struct pipe_screen *pscreen = pipe_loader_create_screen(dev, false); + if (!pscreen) { + fprintf(stderr, "tva: llvmpipe screen creation failed\n"); + pipe_loader_release(&dev, 1); + return NULL; + } + fprintf(stderr, "tva: using the llvmpipe software backend\n"); + return tva_vscreen_from_pscreen(pscreen, dev); +} + +struct vl_screen * +tva_bridge_vscreen_create(int fd, bool honor_dri_prime) +{ + const char *backend = os_get_option("TERMUX_VA_GPU_BACKEND"); + int opened_fd = -1; + if (!backend || !*backend || !strcmp(backend, "auto")) + backend = "auto"; + +#if defined(__linux__) + /* An X11 VA display in a PRoot container does not carry a DRM fd. Open + * KGSL here so the bridge can still create a GPU screen without a DRM + * render node; the loader override below selects the KGSL alias. */ + if (fd < 0 && (!strcmp(backend, "auto") || !strcmp(backend, "kgsl"))) { + opened_fd = open("/dev/kgsl-3d0", O_RDWR | O_CLOEXEC); + if (opened_fd >= 0) { + fd = opened_fd; + if (!strcmp(backend, "auto")) + backend = "kgsl"; + } + } +#endif + + if (getenv("DMD_VA_LOG")) + fprintf(stderr, "tva: creating the bridge vscreen, backend='%s'\n", backend); + + if (!strcmp(backend, "auto") || !strcmp(backend, "kgsl")) { + /* On the kgsl stack the display controller's DRM node drives no GPU; + * without this the freedreno device layer builds a half-initialised + * device that fails at the first pipe query. KGSL cannot export its + * native allocations as dma-buf, so make shareable video surfaces use + * the dma-heap import path before the screen is created. On a real + * msm GPU render node an absent /dev/kgsl-3d0 simply falls through to + * the regular msm path. */ + setenv("FD_FORCE_KGSL", "1", 0); + setenv("FD_KGSL_ENABLE_DMABUF", "1", 0); + } + + if (!strcmp(backend, "auto")) { + /* 1. stock selection (correct on normal GPU render nodes). The kgsl + * stack fails here by construction: the display node's kernel name + * matches no descriptor and zink has no Vulkan device. */ + struct vl_screen *vscreen = vl_drm_screen_create(fd, honor_dri_prime); + if (vscreen) { + if (opened_fd >= 0) + close(opened_fd); + return vscreen; + } + /* 2. software fallback. The KGSL alias is intentionally explicit: + * applications must opt into GPU submission through /dev/kgsl-3d0 + * with TERMUX_VA_GPU_BACKEND=kgsl. */ + fprintf(stderr, "tva: stock drm screen creation failed, using llvmpipe\n"); + struct vl_screen *vscreen_sw = tva_vscreen_sw(); + if (opened_fd >= 0) + close(opened_fd); + return vscreen_sw; + } + if (!strcmp(backend, "kgsl")) { + /* Probe with the loader override so the device resolves to the + * fork's kgsl freedreno alias, then restore the env. */ + const char *old = getenv("MESA_LOADER_DRIVER_OVERRIDE"); + char *saved = old && *old ? strdup(old) : NULL; + setenv("MESA_LOADER_DRIVER_OVERRIDE", "kgsl", 1); + setenv("FD_FORCE_KGSL", "1", 0); + setenv("FD_KGSL_ENABLE_DMABUF", "1", 0); + struct vl_screen *vscreen = vl_drm_screen_create(fd, honor_dri_prime); + if (saved) { + setenv("MESA_LOADER_DRIVER_OVERRIDE", saved, 1); + free(saved); + } else { + unsetenv("MESA_LOADER_DRIVER_OVERRIDE"); + } + if (vscreen) + { + if (opened_fd >= 0) + close(opened_fd); + return vscreen; + } + fprintf(stderr, "tva: kgsl screen creation failed, trying llvmpipe\n"); + struct vl_screen *vscreen_sw = tva_vscreen_sw(); + if (opened_fd >= 0) + close(opened_fd); + return vscreen_sw; + } + if (!strcmp(backend, "sw")) { + struct vl_screen *vscreen = tva_vscreen_sw(); + if (opened_fd >= 0) + close(opened_fd); + return vscreen; + } + if (!strcmp(backend, "drm")) { + struct vl_screen *vscreen = vl_drm_screen_create(fd, honor_dri_prime); + if (opened_fd >= 0) + close(opened_fd); + return vscreen; + } + + fprintf(stderr, "tva: unknown TERMUX_VA_GPU_BACKEND '%s', using auto\n", backend); + if (opened_fd >= 0) + close(opened_fd); + return tva_bridge_vscreen_create(fd, honor_dri_prime); +} diff --git a/src/gallium/frontends/va/tva_bridge.h b/src/gallium/frontends/va/tva_bridge.h new file mode 100644 index 000000000000..0b3812da58bd --- /dev/null +++ b/src/gallium/frontends/va/tva_bridge.h @@ -0,0 +1,79 @@ +/* + * SPDX-License-Identifier: GPL-3.0-or-later + * + * tva_bridge.h - glue between the Mesa VA frontend and the termux-va + * daemon (Termux MediaCodec hardware decode). + * + * Copyright (C) 2026 lfdevs + * + * This program is free software: you can redistribute it and/or modify + * it under the terms of the GNU General Public License as published by + * the Free Software Foundation, version 3 of the License. + * + * This is new code written for the termux-va project (GPL-3.0); the wire + * protocol client it drives is a port of droidspaces-media-decode's + * vaapi-driver/src/dmd_client.c (Apache-2.0, see tva_client.c). + */ +#ifndef TVA_BRIDGE_H +#define TVA_BRIDGE_H + +#include + +struct vl_screen; +struct pipe_context; +struct pipe_screen; + +#ifdef __cplusplus +extern "C" { +#endif + +/* + * Whether the bridge should take over VA decoding in this process. + * + * Activation order: + * - TERMUX_VA_BRIDGE=0 -> never + * - TERMUX_VA_BRIDGE=1 (or true) -> always + * - otherwise: active when TERMUX_VA_SOCKET / TERMUX_VA_SOCKET_DIR is set, + * or when the default endpoint exists as a socket + * (/tmp/termux-va/termux-va.sock) + */ +bool tva_bridge_active(void); + +/* + * Fill the underlying screen's NULL video capability hooks + * (get_video_param / is_video_format_supported) with the bridge's codec + * table. Drivers without a video path (freedreno, llvmpipe) leave them + * NULL, which would fail the VA frontend's init check. Hooks that are + * already present are left untouched. + */ +void tva_bridge_screen_set_video_hooks(struct pipe_screen *screen); + +/* + * Fill the multimedia context's create_video_codec / + * create_video_buffer(_with_modifiers) hooks with the bridge + * implementations. `pipe` must be the REAL context created from the raw + * screen (the implementations receive it as-is; no wrapper object is + * involved). + */ +void tva_bridge_pipe_set_codec_hooks(struct pipe_context *pipe); + +/* + * Create the underlying screen for the VA frontend's vscreen, with backend + * selection (see TERMUX_VA_GPU_BACKEND in docs/envvars.rst): + * + * auto (default) stock loader selection, then llvmpipe + * kgsl force the "kgsl" freedreno alias (GPU submission via + * /dev/kgsl-3d0, handed fd as control/identity fd) + * drm stock loader selection only + * sw llvmpipe only + * + * Does not take ownership of `fd` (pipe_loader dups it internally). + * Returns NULL when every selected backend failed. + */ +struct vl_screen *tva_bridge_vscreen_create(int fd, bool honor_dri_prime); + +#ifdef __cplusplus +} +#endif + +#endif /* TVA_BRIDGE_H */ diff --git a/src/gallium/frontends/va/tva_client.c b/src/gallium/frontends/va/tva_client.c new file mode 100644 index 000000000000..5ccaff3df805 --- /dev/null +++ b/src/gallium/frontends/va/tva_client.c @@ -0,0 +1,1135 @@ +/* + * SPDX-License-Identifier: GPL-3.0-or-later + * + * tva_client.c - termux-va daemon client library implementation. + * + * Copyright (C) 2026 lfdevs + * + * This program is free software: you can redistribute it and/or modify + * it under the terms of the GNU General Public License as published by + * the Free Software Foundation, version 3 of the License. + * + * ****************************************************************************** + * MODIFICATION NOTICE (GPL-3.0 section 5) + * + * This file is a MODIFIED version of vaapi-driver/src/dmd_client.c from the + * droidspaces-media-decode project (Apache License, Version 2.0), ported to + * the termux-va project and relicensed under GPL-3.0. Modifications: + * - the TCP transport was removed entirely: the control channel is always + * a path-based Unix socket, and the "TCP" framing mode is called inline + * (wire value 0 unchanged), + * - protocol constants come from tva_protocol.h (mirror of the daemon's + * common/tva_protocol.h; values unchanged), + * - symbols renamed dmd_* -> tva_*; logs translated to English, + * - endpoint defaults resolve in container view via tva_default_endpoint() + * (TERMUX_VA_SOCKET > TERMUX_VA_SOCKET_DIR > shared-tmp paths) using + * Mesa's os_get_option (which also consults Android system properties), + * - the SHM slot release deadline comment updated for the daemon's 15s + * slot wait. + * Everything else - the v3 handshake with the endpoint inode reconciliation, + * the v2 downgrade retry, the SHM attach via abstract socket + SCM_RIGHTS, + * the format-block handling - is kept faithful to the original. + * ****************************************************************************** + * + * Implementation decisions inherited from the upstream library: + * + * 1) All fds non-blocking + poll: the host is a browser; unbounded blocking + * anywhere would hang the whole process. connect() also goes through + * non-blocking + POLLOUT + SO_ERROR. + * 2) recv/send always loop to completion, handling EINTR and short + * transfers; sends carry MSG_NOSIGNAL. + * 3) SHM is only an INTENT: the daemon sends the handshake response BEFORE + * the memfd handoff and silently downgrades to inline framing if the + * handoff times out. A failed pickup is therefore not an error - just + * continue with inline framing. The fallback is mandatory here. + * 4) Error propagation: every failure path records code + human-readable + * reason (including strerror); callers read tva_session_last_error(). + * 5) No global mutable state: even the log flag lives per session. + */ +#ifndef _GNU_SOURCE +#define _GNU_SOURCE /* MSG_CMSG_CLOEXEC / SOCK_CLOEXEC */ +#endif +#include "tva_client.h" + +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include + +#include "util/os_misc.h" + +struct tva_session { + int fd; /* daemon connection, non-blocking */ + int log; /* log flag, from DMD_VA_LOG */ + + int codec; + int io_timeout_ms; + + char ep_path[108]; /* path used for connect(); stat'ed after the + * handshake for the dev/ino reconciliation */ + + int xfer; /* effective transport (TVA_XFER_*) */ + int eos; /* peer close observed */ + int tx_broken; /* uplink corrupted (send timed out mid-unit) */ + volatile int cancelled; /* abortive cancel requested by the owner */ + + struct tva_format fmt; + struct tva_error err; + + /* Inline-mode receive buffer (reused, grown on demand) */ + uint8_t *rbuf; + size_t rbuf_size; + int rbuf_busy; /* 1 = a frame is outstanding against rbuf */ + + /* SHM pool */ + uint8_t *shm_base; + size_t shm_slot_bytes; + size_t shm_total; + int shm_slots; + int shm_held; /* outstanding slot count, diagnostics */ + + uint64_t units_sent; + uint64_t frames_recv; + uint32_t caps; /* daemon capability bits from the format block */ + uint32_t last_pts; /* unit index of the latest frame, 0 = none */ +}; + +/* ------------------------------------------------------------ logging */ +/* + * Silent by default; DMD_VA_LOG=1 enables it. stderr only. + */ +static void tva_c_log(const struct tva_session *s, const char *fmt, ...) + __attribute__((format(printf, 2, 3))); + +static void tva_c_log(const struct tva_session *s, const char *fmt, ...) +{ + if (!s || !s->log) + return; + va_list ap; + va_start(ap, fmt); + fputs("[tva-client] ", stderr); + vfprintf(stderr, fmt, ap); + fputc('\n', stderr); + va_end(ap); +} + +static int tva_log_wanted(void) +{ + const char *env = getenv("DMD_VA_LOG"); + return (env && env[0] == '1') ? 1 : 0; +} + +/* ------------------------------------------------------------ errors */ +static void err_set(struct tva_error *e, int code, int hs_status, + const char *what, int use_errno) +{ + if (!e) + return; + e->code = code; + e->handshake_status = hs_status; + if (use_errno) { + int sv = errno; + snprintf(e->msg, sizeof(e->msg), "%s: %s", what, strerror(sv)); + errno = sv; + } else { + snprintf(e->msg, sizeof(e->msg), "%s", what); + } +} + +static int sess_err(struct tva_session *s, int code, const char *what, + int use_errno) +{ + err_set(&s->err, code, 0, what, use_errno); + if (s->log) + tva_c_log(s, "error(%d): %s", code, s->err.msg); + return code; +} + +/* ------------------------------------------------------------ fd helpers */ +/* All sockets are created with SOCK_CLOEXEC | SOCK_NONBLOCK in one call so + * no fcntl window exists (a fork race would leak fds into host children). */ + +/* Wait until fd is readable/writable. Returns 1 ready, 0 timeout, -1 error + * (errno valid). timeout_ms < 0 is treated as 0 - the library never waits + * without a bound. */ +static int wait_fd(int fd, short events, int timeout_ms) +{ + if (timeout_ms < 0) + timeout_ms = 0; + struct pollfd p; + p.fd = fd; + p.events = events; + for (;;) { + p.revents = 0; + int r = poll(&p, 1, timeout_ms); + if (r < 0) { + if (errno == EINTR) + continue; /* retry with the original timeout on EINTR */ + return -1; + } + if (r == 0) + return 0; + return 1; + } +} + +/* + * Receive exactly len bytes. + * first_timeout_ms: bound for the FIRST byte (can be short, "any frame?") + * rest_timeout_ms: bound per step after the first byte (must not be + * short, or frames get truncated) + * Returns TVA_OK / TVA_EOS (peer closed before any byte) / TVA_ERR_*. + * A peer close mid-message is TVA_ERR_PROTOCOL: a truncated message is not + * a clean end of stream. + */ +static int recv_exact(struct tva_session *s, void *buf, size_t len, + int first_timeout_ms, int rest_timeout_ms) +{ + uint8_t *p = buf; + size_t got = 0; + + while (got < len) { + if (__atomic_load_n(&s->cancelled, __ATOMIC_ACQUIRE)) + return TVA_ERR_CANCELLED; + int to = (got == 0) ? first_timeout_ms : rest_timeout_ms; + int r = wait_fd(s->fd, POLLIN, to); + if (r < 0) + return sess_err(s, TVA_ERR_IO, "poll for readability failed", 1); + if (__atomic_load_n(&s->cancelled, __ATOMIC_ACQUIRE)) + return TVA_ERR_CANCELLED; + if (r == 0) { + if (got == 0) + return TVA_ERR_TIMEOUT; /* clean "nothing yet" */ + return sess_err(s, TVA_ERR_TIMEOUT, "receive timed out mid-message", 0); + } + + ssize_t n = recv(s->fd, p + got, len - got, 0); + if (n < 0) { + if (errno == EINTR || errno == EAGAIN || errno == EWOULDBLOCK) + continue; + return sess_err(s, TVA_ERR_IO, "recv failed", 1); + } + if (n == 0) { + if (got == 0) { + s->eos = 1; + return TVA_EOS; + } + return sess_err(s, TVA_ERR_PROTOCOL, + "peer closed mid-message (message truncated)", 0); + } + got += (size_t)n; + } + return TVA_OK; +} + +/* Send exactly len bytes; MSG_NOSIGNAL keeps a closed peer from SIGPIPE-ing + * the host process. */ +static int send_exact(struct tva_session *s, const void *buf, size_t len, + int timeout_ms) +{ + const uint8_t *p = buf; + size_t sent = 0; + + while (sent < len) { + if (__atomic_load_n(&s->cancelled, __ATOMIC_ACQUIRE)) + return TVA_ERR_CANCELLED; + ssize_t n = send(s->fd, p + sent, len - sent, MSG_NOSIGNAL); + if (n < 0) { + if (errno == EINTR) + continue; + if (errno == EAGAIN || errno == EWOULDBLOCK) { + int r = wait_fd(s->fd, POLLOUT, timeout_ms); + if (__atomic_load_n(&s->cancelled, __ATOMIC_ACQUIRE)) + return TVA_ERR_CANCELLED; + if (r < 0) + return sess_err(s, TVA_ERR_IO, "poll for writability failed", 1); + if (r == 0) + return sess_err(s, TVA_ERR_TIMEOUT, "send timed out", 0); + continue; + } + if (errno == EPIPE || errno == ECONNRESET) + return sess_err(s, TVA_ERR_IO, "connection closed by the daemon", 1); + return sess_err(s, TVA_ERR_IO, "send failed", 1); + } + sent += (size_t)n; + } + return TVA_OK; +} + +/* ------------------------------------------------------------ shared memory */ +static volatile uint32_t *shm_state_word(struct tva_session *s, int idx) +{ + return (volatile uint32_t *)(s->shm_base + (size_t)idx * sizeof(uint32_t)); +} + +/* + * Connect to the daemon's abstract socket and take the memfd via + * SCM_RIGHTS. Returns -1 on failure (caller downgrades to inline framing; + * not a fatal error). + * + * Abstract address layout: sun_path[0] = 0, name from sun_path+1, + * addrlen = offsetof(sun_path) + 1 + strlen(name) - exactly symmetric to + * the daemon's bind(). + */ +static int shm_attach(struct tva_session *s, const char *name) +{ + int sock = socket(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC | SOCK_NONBLOCK, 0); + if (sock < 0) { + tva_c_log(s, "abstract socket creation failed: %s", strerror(errno)); + return -1; + } + + struct sockaddr_un ua; + memset(&ua, 0, sizeof(ua)); + ua.sun_family = AF_UNIX; + size_t nl = strlen(name); + if (nl > sizeof(ua.sun_path) - 2) { + tva_c_log(s, "shared memory name too long: %zu", nl); + close(sock); + return -1; + } + ua.sun_path[0] = 0; + memcpy(ua.sun_path + 1, name, nl); + socklen_t ulen = (socklen_t)(offsetof(struct sockaddr_un, sun_path) + 1 + nl); + + if (connect(sock, (struct sockaddr *)&ua, ulen) < 0) { + if (errno != EINPROGRESS) { + tva_c_log(s, "connect @%s failed: %s", name, strerror(errno)); + close(sock); + return -1; + } + if (wait_fd(sock, POLLOUT, TVA_SHM_ATTACH_MS) != 1) { + tva_c_log(s, "connect @%s timed out", name); + close(sock); + return -1; + } + int se = 0; + socklen_t sl = sizeof(se); + if (getsockopt(sock, SOL_SOCKET, SO_ERROR, &se, &sl) < 0 || se != 0) { + tva_c_log(s, "connect @%s failed: %s", name, strerror(se ? se : errno)); + close(sock); + return -1; + } + } + + /* 12 bytes [slots][slot bytes][pool total] plus the memfd in SCM_RIGHTS. + * The daemon sends it in one sendmsg; ancillary data never splits across + * messages, but recvmsg may still EAGAIN first on a non-blocking fd. */ + uint32_t meta[3]; + struct iovec io; + io.iov_base = meta; + io.iov_len = sizeof(meta); + char cbuf[CMSG_SPACE(sizeof(int))]; + struct msghdr mh; + ssize_t n; + + for (;;) { + if (wait_fd(sock, POLLIN, TVA_SHM_ATTACH_MS) != 1) { + tva_c_log(s, "timed out waiting for the memfd"); + close(sock); + return -1; + } + memset(cbuf, 0, sizeof(cbuf)); + memset(&mh, 0, sizeof(mh)); + mh.msg_iov = &io; + mh.msg_iovlen = 1; + mh.msg_control = cbuf; + mh.msg_controllen = sizeof(cbuf); + n = recvmsg(sock, &mh, MSG_CMSG_CLOEXEC); + if (n < 0 && (errno == EINTR || errno == EAGAIN || errno == EWOULDBLOCK)) + continue; + break; + } + + if (n != (ssize_t)sizeof(meta)) { + tva_c_log(s, "failed to receive shared memory parameters: n=%zd (%s)", n, + n < 0 ? strerror(errno) : "length mismatch"); + close(sock); + return -1; + } + + int mfd = -1; + for (struct cmsghdr *cm = CMSG_FIRSTHDR(&mh); cm; cm = CMSG_NXTHDR(&mh, cm)) { + if (cm->cmsg_level == SOL_SOCKET && cm->cmsg_type == SCM_RIGHTS && + cm->cmsg_len == CMSG_LEN(sizeof(int))) { + memcpy(&mfd, CMSG_DATA(cm), sizeof(int)); + break; + } + } + close(sock); + if (mfd < 0) { + tva_c_log(s, "no memfd in the response"); + return -1; + } + + int slots = (int)ntohl(meta[0]); + size_t slot_sz = (size_t)ntohl(meta[1]); + size_t total = (size_t)ntohl(meta[2]); + + /* Self-check: refuse to read shared memory through inconsistent layout */ + if (slots <= 0 || slots > 64 || slot_sz == 0 || + total < SHM_CTRL_BYTES + slot_sz * (size_t)slots || + (size_t)slots * sizeof(uint32_t) > SHM_CTRL_BYTES) { + tva_c_log(s, "inconsistent shared memory parameters: slots=%d slot=%zu total=%zu", + slots, slot_sz, total); + close(mfd); + return -1; + } + + void *base = mmap(NULL, total, PROT_READ | PROT_WRITE, MAP_SHARED, mfd, 0); + close(mfd); /* fd no longer needed after mmap */ + if (base == MAP_FAILED) { + tva_c_log(s, "mmap of shared memory failed: %s", strerror(errno)); + return -1; + } + + s->shm_base = base; + s->shm_slots = slots; + s->shm_slot_bytes = slot_sz; + s->shm_total = total; + tva_c_log(s, "shared memory attached: %d slots x %zu bytes (total %zu)", + slots, slot_sz, total); + return 0; +} + +/* ------------------------------------------------- connection & handshake */ +/* Non-blocking connect with a bounded wait. */ +static int sock_connect_wait(int fd, const struct sockaddr *sa, socklen_t slen, + const char *what, int timeout_ms, + struct tva_error *err) +{ + if (connect(fd, sa, slen) < 0) { + if (errno != EINPROGRESS) { + char m[96]; + snprintf(m, sizeof(m), "connect %s failed", what); + err_set(err, TVA_ERR_CONNECT, 0, m, 1); + close(fd); + return -1; + } + int r = wait_fd(fd, POLLOUT, timeout_ms); + if (r < 0) { + err_set(err, TVA_ERR_CONNECT, 0, "poll during connect failed", 1); + close(fd); + return -1; + } + if (r == 0) { + err_set(err, TVA_ERR_CONNECT, 0, "connect timed out", 0); + close(fd); + return -1; + } + int se = 0; + socklen_t sl = sizeof(se); + if (getsockopt(fd, SOL_SOCKET, SO_ERROR, &se, &sl) < 0) { + err_set(err, TVA_ERR_CONNECT, 0, "getsockopt(SO_ERROR) failed", 1); + close(fd); + return -1; + } + if (se != 0) { + errno = se; + err_set(err, TVA_ERR_CONNECT, 0, "connect refused", 1); + close(fd); + return -1; + } + } + return 0; +} + +/* Connect to a path-based Unix socket. + * + * The socket lives in the shared tmp directory, so the container and Termux + * see the same file; authentication is plain file permissions. The receive + * buffer must be enlarged explicitly (see the 4MB note) - the AF_UNIX + * default of 224KB cannot hold a single NV12 frame, which once collapsed + * throughput below realtime. */ +static int unix_connect(struct tva_session *s, const char *path, + int timeout_ms, struct tva_error *err) +{ + int fd = socket(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC | SOCK_NONBLOCK, 0); + if (fd < 0) { + err_set(err, TVA_ERR_CONNECT, 0, "failed to create Unix socket", 1); + return -1; + } + + struct sockaddr_un ua; + memset(&ua, 0, sizeof(ua)); + ua.sun_family = AF_UNIX; + if (strlen(path) >= sizeof(ua.sun_path)) { + err_set(err, TVA_ERR_CONNECT, 0, "socket path too long", 0); + close(fd); + return -1; + } + memcpy(ua.sun_path, path, strlen(path)); + + if (sock_connect_wait(fd, (struct sockaddr *)&ua, sizeof(ua), + path, timeout_ms, err) < 0) + return -1; + + { + int bufsz = 4 * 1024 * 1024; + (void)setsockopt(fd, SOL_SOCKET, SO_RCVBUF, &bufsz, sizeof(bufsz)); + (void)setsockopt(fd, SOL_SOCKET, SO_SNDBUF, &bufsz, sizeof(bufsz)); + } + + /* Remember the endpoint path for the post-handshake inode check */ + snprintf(s->ep_path, sizeof(s->ep_path), "%s", path); + s->fd = fd; + return 0; +} + +/* + * Handshake. The request is exactly 24 bytes, all big-endian: + * [4B magic][4B version][4B codec][4B width][4B height][4B xfer] + * The daemon reads 4 bytes of magic first, then the remaining 20 - the + * request must be written in one piece. This client sends version 3. + * + * The response is at least 12 bytes: [status][actual xfer][namelen]. With + * v3 and status==0, bit31 of namelen marks 16 extra bytes of endpoint + * extension (dev/ino split into high/low u32s); the client stats the path + * it connected to and reconciles, failing with TVA_ERR_ENDPOINT_MISMATCH on + * a mismatch. Old daemons reply with a bare 12 bytes -> check skipped. + */ +static int do_handshake(struct tva_session *s, + const struct tva_session_config *cfg, + uint32_t use_version, + struct tva_error *err) +{ + uint32_t hello[6]; + hello[0] = htonl(HELLO_MAGIC); + hello[1] = htonl(use_version); + hello[2] = htonl((uint32_t)cfg->codec); + hello[3] = htonl((uint32_t)cfg->width); + hello[4] = htonl((uint32_t)cfg->height); + hello[5] = htonl(cfg->want_shm ? (uint32_t)XFER_SHM + : (uint32_t)XFER_INLINE); + + if (send_exact(s, hello, sizeof(hello), s->io_timeout_ms) != TVA_OK) { + if (err) + *err = s->err; + return -1; + } + + uint32_t head[3]; + int r = recv_exact(s, head, sizeof(head), s->io_timeout_ms, s->io_timeout_ms); + if (r != TVA_OK) { + if (r == TVA_EOS) + sess_err(s, TVA_ERR_PROTOCOL, + "daemon closed without a handshake response (version mismatch?)", 0); + else if (r == TVA_ERR_TIMEOUT) + sess_err(s, TVA_ERR_TIMEOUT, "timed out waiting for the handshake response", 0); + if (err) + *err = s->err; + return -1; + } + + uint32_t status = ntohl(head[0]); + uint32_t mode = ntohl(head[1]); + uint32_t nlen_w = ntohl(head[2]); + int has_ext = (nlen_w >> 31) != 0; /* v3: bit31 = endpoint extension */ + uint32_t nlen = nlen_w & 0x7fffffffu; + uint64_t ep_dev = 0, ep_ino = 0; + + /* Error paths reply with a bare 12 bytes (nlen=0); handle rejection + * before reading anything else */ + if (status != 0) { + const char *why = (status == 1) ? "daemon rejected handshake: protocol version not supported" + : (status == 2) ? "daemon rejected handshake: codec not supported" + : (status == 3) ? "daemon rejected handshake: resolution outside 96x96..8192x4320" + : (status == 4) ? "daemon rejected handshake: handshake missing" + : "daemon rejected handshake: unknown status"; + err_set(&s->err, TVA_ERR_REJECTED, (int)status, why, 0); + if (err) + *err = s->err; + return -1; + } + + /* v3 endpoint extension: [u32 dev_hi][u32 dev_lo][u32 ino_hi][u32 ino_lo] */ + if (has_ext) { + uint32_t ext[4]; + if (recv_exact(s, ext, sizeof(ext), + s->io_timeout_ms, s->io_timeout_ms) != TVA_OK) { + sess_err(s, TVA_ERR_PROTOCOL, "failed to read the endpoint extension", 0); + if (err) + *err = s->err; + return -1; + } + ep_dev = ((uint64_t)ntohl(ext[0]) << 32) | ntohl(ext[1]); + ep_ino = ((uint64_t)ntohl(ext[2]) << 32) | ntohl(ext[3]); + } + + char name[64]; + memset(name, 0, sizeof(name)); + if (nlen > 0) { + if (nlen >= sizeof(name)) { + sess_err(s, TVA_ERR_PROTOCOL, "illegal name length in the handshake response", 0); + if (err) + *err = s->err; + return -1; + } + if (recv_exact(s, name, nlen, s->io_timeout_ms, s->io_timeout_ms) != TVA_OK) { + sess_err(s, TVA_ERR_PROTOCOL, "failed to read the shared memory name", 0); + if (err) + *err = s->err; + return -1; + } + } + + /* + * Endpoint inode reconciliation - the core of the v3 extension. + * + * connect() succeeding only means "something is listening on that + * path", not that it is the endpoint we mean. If a single socket FILE + * was bind-mounted (instead of a directory) and the daemon restarted, + * the client resolves a stale orphan socket; both sides may even stat + * the SAME orphan inode. The daemon now reports its listening + * endpoint's (st_dev, st_ino) and the client verifies - on mismatch, + * fail immediately with an actionable message instead of running on a + * fake connection. + */ + if (s->ep_path[0] && !(ep_dev == 0 && ep_ino == 0)) { + struct stat st; + if (stat(s->ep_path, &st) != 0) { + char msg[192]; + snprintf(msg, sizeof(msg), + "cannot stat endpoint %s for the inode check, refusing to continue", s->ep_path); + sess_err(s, TVA_ERR_ENDPOINT_MISMATCH, msg, 1); + if (err) + *err = s->err; + return -1; + } + uint64_t my_dev = (uint64_t)st.st_dev, my_ino = (uint64_t)st.st_ino; + if (my_dev != ep_dev || my_ino != ep_ino) { + if (s->log) + tva_c_log(s, "endpoint inode mismatch details: path=%s " + "stat(dev=%llu,ino=%llu) != daemon(dev=%llu,ino=%llu)", + s->ep_path, + (unsigned long long)my_dev, (unsigned long long)my_ino, + (unsigned long long)ep_dev, (unsigned long long)ep_ino); + char msg[192]; + snprintf(msg, sizeof(msg), + "endpoint inode mismatch: stat(ino=%llu) != daemon(ino=%llu)" + "; the mount points at a stale socket (bind-mount the directory," + " not the socket file)", + (unsigned long long)my_ino, (unsigned long long)ep_ino); + sess_err(s, TVA_ERR_ENDPOINT_MISMATCH, msg, 0); + if (err) + *err = s->err; + return -1; + } + if (s->log) + tva_c_log(s, "endpoint check passed: dev=%llu ino=%llu", + (unsigned long long)my_dev, (unsigned long long)my_ino); + } + + /* mode==SHM is only the daemon's intent: the response precedes the memfd + * handoff, and a failed handoff downgrades silently. A failed pickup is + * therefore never fatal - continue with inline framing. */ + s->xfer = XFER_INLINE; + if (mode == (uint32_t)XFER_SHM && nlen > 0) { + if (shm_attach(s, name) == 0) + s->xfer = XFER_SHM; + else + tva_c_log(s, "shared memory pickup failed, continuing with inline framing"); + } + return 0; +} + +/* ------------------------------------------------------------ public API */ +void tva_session_config_defaults(struct tva_session_config *cfg) +{ + if (!cfg) + return; + memset(cfg, 0, sizeof(*cfg)); + cfg->codec = CODEC_H264; + cfg->connect_timeout_ms = TVA_DEF_CONNECT_MS; + cfg->io_timeout_ms = TVA_DEF_IO_MS; +} + +const char *tva_default_endpoint(char *buf, size_t bufsz) +{ + const char *env_sock = os_get_option("TERMUX_VA_SOCKET"); + if (env_sock && *env_sock) + return env_sock; + + const char *env_dir = os_get_option("TERMUX_VA_SOCKET_DIR"); + if (env_dir && *env_dir) { + snprintf(buf, bufsz, "%s/%s", env_dir, TVA_SOCK_NAME); + return buf; + } + + /* The daemon applies a termux-x11-style fallback chain to TMPDIR; + * mirror it, then prefer whichever default path actually exists. */ + const char *tmp = getenv("TMPDIR"); + if (!tmp || !*tmp || !strcmp(tmp, "/data/local/tmp")) + tmp = NULL; + + if (tmp) { + snprintf(buf, bufsz, "%s/%s/%s", tmp, TVA_SOCKET_DIR_NAME, TVA_SOCK_NAME); + struct stat st; + if (stat(buf, &st) == 0 && S_ISSOCK(st.st_mode)) + return buf; + } + snprintf(buf, bufsz, "/tmp/%s/%s", TVA_SOCKET_DIR_NAME, TVA_SOCK_NAME); + return buf; +} + +int tva_format_display_width(const struct tva_format *fmt) +{ + if (!fmt || !fmt->valid) + return 0; + int w = fmt->crop_right - fmt->crop_left + 1; + return w > 0 ? w : 0; +} + +int tva_format_display_height(const struct tva_format *fmt) +{ + if (!fmt || !fmt->valid) + return 0; + int h = fmt->crop_bottom - fmt->crop_top + 1; + return h > 0 ? h : 0; +} + +struct tva_session *tva_session_create(const struct tva_session_config *cfg, + struct tva_error *err) +{ + if (err) { + memset(err, 0, sizeof(*err)); + } + if (!cfg) { + err_set(err, TVA_ERR_INVAL, 0, "cfg is NULL", 0); + return NULL; + } + if (cfg->codec < CODEC_H264 || cfg->codec >= CODEC_MAX) { + err_set(err, TVA_ERR_INVAL, 0, "invalid codec id", 0); + return NULL; + } + if (cfg->width < 96 || cfg->height < 96 || + cfg->width > 8192 || cfg->height > 4320) { + err_set(err, TVA_ERR_INVAL, 0, + "resolution outside the daemon range 96x96..8192x4320", 0); + return NULL; + } + + struct tva_session *s = calloc(1, sizeof(*s)); + if (!s) { + err_set(err, TVA_ERR_NOMEM, 0, "session allocation failed", 0); + return NULL; + } + s->fd = -1; + s->log = tva_log_wanted(); + s->codec = cfg->codec; + s->io_timeout_ms = cfg->io_timeout_ms > 0 ? cfg->io_timeout_ms : TVA_DEF_IO_MS; + s->xfer = XFER_INLINE; + + int cto = cfg->connect_timeout_ms > 0 ? cfg->connect_timeout_ms + : TVA_DEF_CONNECT_MS; + + char defep[300]; + const char *path = cfg->sock_path ? cfg->sock_path + : tva_default_endpoint(defep, sizeof(defep)); + fprintf(stderr, "tva-client: opening %s codec=%d %dx%d shm=%d\n", + path ? path : "(null)", cfg->codec, cfg->width, cfg->height, + cfg->want_shm); + if (unix_connect(s, path, cto, err) < 0) { + fprintf(stderr, "tva-client: connect failed code=%d msg=%s\n", + err ? err->code : 0, err && err->msg[0] ? err->msg : "(none)"); + if (err) + s->err = *err; + tva_session_destroy(s); + return NULL; + } + if (do_handshake(s, cfg, HELLO_VERSION, err) < 0) { + fprintf(stderr, "tva-client: handshake failed code=%d status=%d msg=%s\n", + s->err.code, s->err.handshake_status, + s->err.msg[0] ? s->err.msg : "(none)"); + /* + * Version downgrade retry: daemons that check the version strictly + * reject v3 with status=1. Retry once with v2 (whose response has + * no endpoint extension; the inode check is skipped). Only a + * version rejection downgrades - codec/resolution rejections are + * unaffected by the version. + */ + if (s->err.code == TVA_ERR_REJECTED && s->err.handshake_status == 1 && + HELLO_VERSION > TVA_VERSION_MIN) { + tva_c_log(s, "daemon does not accept protocol v%u, retrying with v2" + " (inode check will be skipped)", + (unsigned)HELLO_VERSION); + if (s->fd >= 0) { + close(s->fd); /* the daemon closed it; reconnect */ + s->fd = -1; + } + s->ep_path[0] = '\0'; + memset(&s->err, 0, sizeof(s->err)); + if (unix_connect(s, path, cto, err) < 0 || + do_handshake(s, cfg, TVA_VERSION_MIN, err) < 0) { + if (err) + s->err = *err; + tva_session_destroy(s); + return NULL; + } + } else { + if (err) + s->err = s->err; + tva_session_destroy(s); + return NULL; + } + } + + tva_c_log(s, "session established: unix=%s codec=%d %dx%d frame-delivery=%s", + s->ep_path, cfg->codec, cfg->width, cfg->height, + s->xfer == XFER_SHM ? "SHM" : "inline"); + return s; +} + +void tva_session_cancel(struct tva_session *s) +{ + if (!s) + return; + __atomic_store_n(&s->cancelled, 1, __ATOMIC_RELEASE); + if (s->fd >= 0) + shutdown(s->fd, SHUT_RDWR); +} + +void tva_session_destroy(struct tva_session *s) +{ + if (!s) + return; + if (s->shm_base) { + /* Return every still-held slot: the daemon may still be waiting for + * them and holding slots makes it wait out its timeout. */ + for (int i = 0; i < s->shm_slots; i++) + __atomic_store_n(shm_state_word(s, i), 0u, __ATOMIC_RELEASE); + munmap(s->shm_base, s->shm_total); + s->shm_base = NULL; + } + if (s->fd >= 0) { + close(s->fd); + s->fd = -1; + } + free(s->rbuf); + s->rbuf = NULL; + free(s); +} + +int tva_session_send_unit(struct tva_session *s, const void *data, size_t len) +{ + if (!s) + return TVA_ERR_INVAL; + if (!data || len == 0) + return sess_err(s, TVA_ERR_INVAL, "empty data unit", 0); + if (len > TVA_MAX_UNIT_BYTES) + return sess_err(s, TVA_ERR_TOOBIG, + "data unit exceeds the daemon's 8MB cap", 0); + if (s->fd < 0) + return sess_err(s, TVA_ERR_STATE, "session has no live connection", 0); + /* A previous send failed after the length prefix went out: the uplink + * byte stream is misaligned and further sends would feed the daemon + * data as lengths. Refuse and force a session rebuild. */ + if (s->tx_broken) + return sess_err(s, TVA_ERR_STATE, + "uplink corrupted (earlier send interrupted); rebuild the session", 0); + + /* H.264/HEVC units MUST carry an Annex B start code: the daemon locates + * the nal_unit_header through it to recognize SPS/PPS/VPS. Missing + * start codes make the decoder silently swallow everything and produce + * no frames - fail here instead. VP8/VP9 is the opposite: adding start + * codes would corrupt the frame, so none are ever added. */ + if (s->codec == CODEC_H264 || s->codec == CODEC_HEVC) { + const uint8_t *b = data; + int sc3 = (len >= 3 && b[0] == 0 && b[1] == 0 && b[2] == 1); + int sc4 = (len >= 4 && b[0] == 0 && b[1] == 0 && b[2] == 0 && b[3] == 1); + if (!sc3 && !sc4) + return sess_err(s, TVA_ERR_PROTOCOL, + "H.264/HEVC data unit lacks an Annex B start code", 0); + } + + uint32_t be = htonl((uint32_t)len); + int r = send_exact(s, &be, 4, s->io_timeout_ms); + if (r != TVA_OK) { + /* the length prefix itself may be partially written: stream broken */ + s->tx_broken = 1; + return r; + } + r = send_exact(s, data, len, s->io_timeout_ms); + if (r != TVA_OK) { + s->tx_broken = 1; + return r; + } + + s->units_sent++; + return TVA_OK; +} + +static int ensure_rbuf(struct tva_session *s, size_t need) +{ + if (s->rbuf_size >= need) + return TVA_OK; + size_t ns = need + need / 2; + if (ns < 256 * 1024) + ns = 256 * 1024; + uint8_t *nb = realloc(s->rbuf, ns); + if (!nb) + return sess_err(s, TVA_ERR_NOMEM, "failed to grow the receive buffer", 0); + s->rbuf = nb; + s->rbuf_size = ns; + return TVA_OK; +} + +/* Stamp the current format snapshot onto the frame so every frame carries + * self-consistent geometry. */ +static void frame_apply_format(const struct tva_session *s, struct tva_frame *f) +{ + f->unit_seq = s->last_pts; + f->stride = s->fmt.stride; + f->slice_height = s->fmt.slice_height; + f->crop_left = s->fmt.crop_left; + f->crop_top = s->fmt.crop_top; + f->crop_right = s->fmt.crop_right; + f->crop_bottom = s->fmt.crop_bottom; +} + +int tva_session_next_frame(struct tva_session *s, struct tva_frame *out, + int timeout_ms) +{ + if (!s) + return TVA_ERR_INVAL; + if (__atomic_load_n(&s->cancelled, __ATOMIC_ACQUIRE)) + return TVA_ERR_CANCELLED; + if (!out) + return sess_err(s, TVA_ERR_INVAL, "out is NULL", 0); + if (s->fd < 0) + return sess_err(s, TVA_ERR_STATE, "session has no live connection", 0); + if (s->eos) + return TVA_EOS; + if (s->rbuf_busy) + return sess_err(s, TVA_ERR_STATE, + "previous frame not released, the receive buffer is still held", 0); + + int first_to = (timeout_ms < 0) ? s->io_timeout_ms : timeout_ms; + int rest_to = s->io_timeout_ms; + + /* Loop: the header may introduce a format block; consume it and keep + * waiting for a real frame */ + for (;;) { + uint8_t hdr[12]; + int r = recv_exact(s, hdr, sizeof(hdr), first_to, rest_to); + if (r != TVA_OK) + return r; /* TVA_EOS / TIMEOUT / errors as-is */ + + /* byte-wise memcpy + ntohl: hdr is uint8_t[]; a direct uint32_t* + * dereference may be unaligned on aarch64 (undefined behavior) */ + uint32_t w_be, h_be, sz_be; + memcpy(&w_be, hdr + 0, 4); + memcpy(&h_be, hdr + 4, 4); + memcpy(&sz_be, hdr + 8, 4); + uint32_t w = ntohl(w_be), h = ntohl(h_be), sz = ntohl(sz_be); + + if (sz == FMTDESC_SENTINEL) { + /* word 2 is the capability flags (0 on legacy daemons) */ + s->caps = h; + /* [0][caps][0xFFFFFFFF] is followed by 8 big-endian words */ + uint32_t fw[FMTDESC_WORDS]; + r = recv_exact(s, fw, sizeof(fw), rest_to, rest_to); + if (r == TVA_EOS) + return sess_err(s, TVA_ERR_PROTOCOL, "format block truncated", 0); + if (r != TVA_OK) + return r; + s->fmt.buf_width = (int)ntohl(fw[0]); + s->fmt.buf_height = (int)ntohl(fw[1]); + s->fmt.stride = (int)ntohl(fw[2]); + s->fmt.slice_height = (int)ntohl(fw[3]); + s->fmt.crop_left = (int)ntohl(fw[4]); + s->fmt.crop_top = (int)ntohl(fw[5]); + s->fmt.crop_right = (int)ntohl(fw[6]); + s->fmt.crop_bottom = (int)ntohl(fw[7]); + s->fmt.valid = 1; + s->fmt.changes++; + tva_c_log(s, "format block #%d: buffer %dx%d stride=%d slice=%d display %dx%d", + s->fmt.changes, s->fmt.buf_width, s->fmt.buf_height, + s->fmt.stride, s->fmt.slice_height, + tva_format_display_width(&s->fmt), + tva_format_display_height(&s->fmt)); + /* bytes already arrived: keep using rest_to so a probe-style + * short timeout cannot misjudge a half-read block */ + first_to = rest_to; + continue; + } + + if (sz == SHMFRAME_SENTINEL) { + /* SHM control message: [slot][length] follow the 12-byte head, + * plus the unit-index word when the daemon announced PTS */ + uint32_t si[2]; + r = recv_exact(s, si, sizeof(si), rest_to, rest_to); + if (r == TVA_EOS) + return sess_err(s, TVA_ERR_PROTOCOL, "SHM control message truncated", 0); + if (r != TVA_OK) + return r; + int slot = (int)ntohl(si[0]); + uint32_t dlen = ntohl(si[1]); + + if (s->caps & CAP_FRAME_PTS) { + uint32_t p_be; + r = recv_exact(s, &p_be, 4, rest_to, rest_to); + if (r == TVA_EOS) + return sess_err(s, TVA_ERR_PROTOCOL, + "SHM message PTS field truncated", 0); + if (r != TVA_OK) + return r; + s->last_pts = ntohl(p_be); + } else { + s->last_pts = 0; + } + + if (!s->shm_base) + return sess_err(s, TVA_ERR_PROTOCOL, + "SHM frame received but shared memory not attached", 0); + if (slot < 0 || slot >= s->shm_slots) + return sess_err(s, TVA_ERR_PROTOCOL, "SHM slot out of range", 0); + if ((size_t)dlen > s->shm_slot_bytes) + return sess_err(s, TVA_ERR_PROTOCOL, "SHM frame exceeds the slot size", 0); + + memset(out, 0, sizeof(*out)); + out->data = s->shm_base + SHM_CTRL_BYTES + + (size_t)slot * s->shm_slot_bytes; + out->size = dlen; + out->width = w; + out->height = h; + out->unit_seq = s->last_pts; + out->shm_slot = slot; + out->seq = s->frames_recv; + frame_apply_format(s, out); + s->frames_recv++; + s->shm_held++; + return TVA_OK; + } + + /* Plain inline frame. A daemon with CAP_FRAME_PTS sends one extra + * word after the 12-byte header: the input unit index. It MUST be + * consumed before the frame body or the stream misaligns. */ + if (s->caps & CAP_FRAME_PTS) { + uint32_t p_be; + r = recv_exact(s, &p_be, 4, first_to, rest_to); + if (r == TVA_EOS) + return sess_err(s, TVA_ERR_PROTOCOL, "frame header PTS field truncated", 0); + if (r != TVA_OK) + return r; + s->last_pts = ntohl(p_be); + } else { + s->last_pts = 0; /* 0 = no PTS info */ + } + + if (sz == 0) + return sess_err(s, TVA_ERR_PROTOCOL, "frame length is 0", 0); + if (sz > TVA_MAX_FRAME_BYTES || w > 16384 || h > 16384) + return sess_err(s, TVA_ERR_PROTOCOL, "unreasonable frame header values", 0); + + r = ensure_rbuf(s, sz); + if (r != TVA_OK) + return r; + r = recv_exact(s, s->rbuf, sz, rest_to, rest_to); + if (r == TVA_EOS) + return sess_err(s, TVA_ERR_PROTOCOL, "frame data truncated", 0); + if (r != TVA_OK) + return r; + + memset(out, 0, sizeof(*out)); + out->data = s->rbuf; + out->size = sz; + out->width = w; + out->height = h; + out->shm_slot = -1; + out->seq = s->frames_recv; + frame_apply_format(s, out); + s->frames_recv++; + s->rbuf_busy = 1; + return TVA_OK; + } +} + +int tva_session_drain(struct tva_session *s) +{ + if (!s) + return TVA_ERR_INVAL; + if (s->fd < 0) + return sess_err(s, TVA_ERR_STATE, "session has no live connection", 0); + if (s->tx_broken) + return sess_err(s, TVA_ERR_STATE, + "uplink corrupted (earlier send interrupted); rebuild the session", 0); + + /* length 0 = reversible drain: the daemon queues EOS, flushes the + * decoder, re-sends the CSD and the session stays alive. */ + uint32_t be = htonl(0); + int r = send_exact(s, &be, 4, s->io_timeout_ms); + if (r != TVA_OK) { + s->tx_broken = 1; + return r; + } + return TVA_OK; +} + +int tva_session_release_frame(struct tva_session *s, struct tva_frame *f) +{ + if (!s) + return TVA_ERR_INVAL; + if (!f || !f->data) + return TVA_OK; /* unconditional release is safe */ + + if (f->shm_slot >= 0) { + if (!s->shm_base || f->shm_slot >= s->shm_slots) + return sess_err(s, TVA_ERR_INVAL, "invalid slot in release", 0); + /* release ordering: all reads of the frame data happen before the + * daemon sees the state word reset and overwrites the slot */ + __atomic_store_n(shm_state_word(s, f->shm_slot), 0u, __ATOMIC_RELEASE); + if (s->shm_held > 0) + s->shm_held--; + } else { + s->rbuf_busy = 0; + } + + f->data = NULL; + f->size = 0; + f->shm_slot = -1; + return TVA_OK; +} + +const struct tva_format *tva_session_format(const struct tva_session *s) +{ + return s ? &s->fmt : NULL; +} + +const char *tva_session_last_error(const struct tva_session *s) +{ + if (!s) + return "session is NULL"; + return s->err.msg; +} + +int tva_session_last_error_code(const struct tva_session *s) +{ + return s ? s->err.code : TVA_ERR_INVAL; +} + +int tva_session_xfer_mode(const struct tva_session *s) +{ + return s ? s->xfer : XFER_INLINE; +} + +uint64_t tva_session_units_sent(const struct tva_session *s) +{ + return s ? s->units_sent : 0; +} + +uint64_t tva_session_frames_received(const struct tva_session *s) +{ + return s ? s->frames_recv : 0; +} diff --git a/src/gallium/frontends/va/tva_client.h b/src/gallium/frontends/va/tva_client.h new file mode 100644 index 000000000000..7daf0c8d5d4e --- /dev/null +++ b/src/gallium/frontends/va/tva_client.h @@ -0,0 +1,277 @@ +/* + * SPDX-License-Identifier: GPL-3.0-or-later + * + * tva_client.h - termux-va daemon client library (used by the Mesa + * termux-va VA bridge). + * + * Copyright (C) 2026 lfdevs + * + * This program is free software: you can redistribute it and/or modify + * it under the terms of the GNU General Public License as published by + * the Free Software Foundation, version 3 of the License. + * + * ****************************************************************************** + * MODIFICATION NOTICE (GPL-3.0 section 5) + * + * This file is a MODIFIED version of vaapi-driver/src/dmd_client.h from the + * droidspaces-media-decode project (Apache License, Version 2.0), ported to + * the termux-va project and relicensed under GPL-3.0. Modifications: + * - the transport is a path-based Unix socket only (the TCP fallback and + * the TCP framing mode were removed; wire values unchanged), + * - protocol constants come from tva_protocol.h (the cmp-verified mirror + * of the daemon's copy), + * - symbols renamed dmd_* -> tva_*, + * - endpoint resolution defaults to /tmp/termux-va/termux-va.sock + * (the shared-tmp view of the Termux daemon's $TMPDIR/termux-va/), + * overridable through TERMUX_VA_SOCKET / TERMUX_VA_SOCKET_DIR. + * ****************************************************************************** + * + * This library wraps "being a client of the Android decode daemon" into an + * opaque session handle. It is compiled into libgallium and dlopen'ed into + * Firefox / ffmpeg / Chrome processes, so it keeps the hard constraints of + * the upstream library it derives from: + * + * - never exit()/abort()/assert(): every error is reported via return code + * - never write stdout; logs go to stderr, silent unless DMD_VA_LOG=1 + * - no global mutable state: all state lives in the session struct, + * multiple sessions may be open concurrently in one process + * - every blocking operation has a timeout (poll + non-blocking fds) + * - sends use MSG_NOSIGNAL: a closed peer cannot SIGPIPE the host + * - all fds are CLOEXEC: the host may fork/exec without leaking fds + * - depends on libc only (plus Mesa's os_get_option for the endpoint) + */ +#ifndef TVA_CLIENT_H +#define TVA_CLIENT_H + +#include +#include + +#include "tva_protocol.h" + +#ifdef __cplusplus +extern "C" { +#endif + +/* ----------------------------------------------------------- return codes */ +/* + * 0 success, positive non-error stream states, negative errors. + * dmd-style legacy names are kept for the error codes the rest of the + * protocol tooling refers to. + */ +enum { + TVA_OK = 0, + TVA_EOS = 1, /* peer closed and no more frames will come */ + + TVA_ERR_INVAL = -1, + TVA_ERR_NOMEM = -2, + TVA_ERR_CONNECT = -3, /* connect failed or timed out */ + TVA_ERR_REJECTED = -4, /* daemon rejected the handshake, see error hs_status */ + TVA_ERR_IO = -5, /* socket read/write failed */ + TVA_ERR_TIMEOUT = -6, /* wait timed out */ + TVA_ERR_PROTOCOL = -7, /* bytes not conforming to the protocol */ + TVA_ERR_STATE = -8, /* session state does not allow the operation */ + TVA_ERR_TOOBIG = -9, /* unit/frame beyond the protocol cap */ + TVA_ERR_ENDPOINT_MISMATCH = -10, + TVA_ERR_CANCELLED = -11 /* tva_session_cancel() interrupted the session */ + /* The (dev,ino) the client stat'ed differs from what the daemon + * reported. Typical cause: a single socket FILE (not a directory) was + * bind-mounted and the daemon restarted, changing the inode. Not + * retryable, not downgradeable - fail loudly. */ +}; + +/* Uplink unit cap (MAX_FRAME; the daemon rejects anything larger). */ +#define TVA_MAX_UNIT_BYTES (8u * 1024u * 1024u) +/* Rational cap of a downlink frame. MUST be far above TVA_MAX_UNIT_BYTES: + * a single 4K NV12 frame is 12441600 bytes; the 8MB cap only bounds uplink. */ +#define TVA_MAX_FRAME_BYTES (64u * 1024u * 1024u) + +/* Default connect / io timeouts (ms). */ +#define TVA_DEF_CONNECT_MS 2000 +#define TVA_DEF_IO_MS 5000 +/* Window to pick up the memfd. The daemon waits 3s; 1.5s here keeps both + * sides' views consistent. */ +#define TVA_SHM_ATTACH_MS 1500 + +/* ------------------------------------------------------------ configuration */ +struct tva_session_config { + /* Path of the Unix socket to connect to. NULL = use the default + * resolution (tva_default_endpoint()). */ + const char *sock_path; + int codec; /* CODEC_* from tva_protocol.h */ + int width; /* 96..8192 */ + int height; /* 96..4320 */ + int want_shm; /* non-0 = request SHM (daemon may downgrade) */ + int connect_timeout_ms;/* <=0 = TVA_DEF_CONNECT_MS */ + int io_timeout_ms; /* <=0 = TVA_DEF_IO_MS */ +}; + +/* Fill cfg with defaults (H.264, inline, default timeouts). */ +void tva_session_config_defaults(struct tva_session_config *cfg); + +/* + * Resolve the default endpoint the same way the daemon does, in container + * view: + * 1. TERMUX_VA_SOCKET (full socket file path) + * 2. TERMUX_VA_SOCKET_DIR (directory + TVA_SOCK_NAME) + * 3. $TMPDIR/termux-va/termux-va.sock (TMPDIR==/data/local/tmp ignored) + * 4. /tmp/termux-va/termux-va.sock + * When several candidates exist, the first that exists as a socket wins; + * otherwise the last candidate is returned so connect() reports the error. + * Returns a pointer into `buf`. + */ +const char *tva_default_endpoint(char *buf, size_t bufsz); + +/* ------------------------------------------------------------ error detail */ +struct tva_error { + int code; /* TVA_ERR_* */ + int handshake_status; /* code==TVA_ERR_REJECTED: daemon status + * 1=version 2=codec 3=resolution 4=missing hello */ + char msg[192]; /* always NUL-terminated */ +}; + +/* ------------------------------------------------------------ format */ +/* + * Geometry of the decoder output buffer, from the daemon's format block. + * + * buf_width/buf_height are the PADDED buffer dimensions (Qualcomm Venus + * aligns width to 128 and height to 32; 1080p decodes into 1920x1088); the + * visible area is the crop rect, which is a CLOSED interval: + * display width = crop_right - crop_left + 1 + * display height = crop_bottom - crop_top + 1 + * The UV plane starts at stride * slice_height. + */ +struct tva_format { + int buf_width; + int buf_height; + int stride; + int slice_height; + int crop_left; + int crop_top; + int crop_right; + int crop_bottom; + int valid; /* 0 = no format block received yet */ + int changes; /* count of format blocks, >1 = mid-stream change */ +}; + +int tva_format_display_width(const struct tva_format *fmt); +int tva_format_display_height(const struct tva_format *fmt); + +/* ------------------------------------------------------------ frame */ +/* + * One NV12 frame. Ownership of data always stays with the library: + * SHM mode -> points into the shared memory slot; release resets the + * slot state word, handing it back to the daemon + * inline -> points into the session receive buffer; release just marks + * it reusable + * data is invalid after release. Forgetting release in SHM mode makes the + * daemon judge the client stuck (~15s slot timeout) - release is mandatory. + */ +struct tva_frame { + uint8_t *data; + size_t size; + + uint32_t width; /* buffer dimensions as announced in the header */ + uint32_t height; + + /* Input unit index (1-based) this frame belongs to; 0 = not provided. + * Used to pair frames with submissions without knowing the decoder's + * output order. */ + uint32_t unit_seq; + + /* Geometry snapshot from the latest format block, self-consistent + * with this frame. */ + int stride; + int slice_height; + int crop_left; + int crop_top; + int crop_right; + int crop_bottom; + + int shm_slot; /* >=0 = SHM slot; -1 = inline receive buffer */ + uint64_t seq; /* session frame counter, starting at 0 */ +}; + +/* ------------------------------------------------------------ session */ +struct tva_session; + +/* + * Create a session: connect to the daemon, handshake, and (if requested) + * take over the shared memory pool. Returns NULL on failure; err (optional) + * receives the reason. No fd or mapping leaks on failure. + */ +struct tva_session *tva_session_create(const struct tva_session_config *cfg, + struct tva_error *err); + +/* + * Interrupt a session's blocking receive/send operations without freeing it. + * The session remains owned by the caller and must still be destroyed after + * any worker using it has joined. + */ +void tva_session_cancel(struct tva_session *s); + +/* Destroy the session: close, unmap, free. NULL is a no-op. */ +void tva_session_destroy(struct tva_session *s); + +/* + * Send one data unit. + * H.264/HEVC: one or more NALUs WITH their Annex B start codes (3 or 4 + * bytes each) - the daemon locates the first nal_unit_header + * through the leading start code. A complete access unit may + * contain multiple slice NALUs. + * VP8/VP9: one whole frame WITHOUT start codes + * The library never adds start codes itself (adding them wrongly corrupts + * the stream silently); H.264/HEVC units without a start code are rejected + * with TVA_ERR_PROTOCOL. Returns TVA_OK / TVA_ERR_*. + */ +int tva_session_send_unit(struct tva_session *s, const void *data, size_t len); + +/* + * Take back a frame. SHM mode returns the slot; inline mode releases the + * receive buffer. Unconditional calls are safe (NULL or released is a + * no-op). + */ +/* + * Reversible drain: send a zero-length unit so the daemon flushes the + * decoder and emits whatever it holds; the session stays usable. The + * reference chain is destroyed by the flush, so frames after a drain + * decode against broken references until the next IDR - send it only when + * waiting is provably futile. + */ +int tva_session_drain(struct tva_session *s); + +/* Take back a frame. SHM mode returns the slot; inline mode releases the + * receive buffer. Unconditional calls are safe (NULL or released is a + * no-op). */ +int tva_session_release_frame(struct tva_session *s, struct tva_frame *f); + +/* + * Fetch the next frame. timeout_ms < 0 uses the configured io timeout. + * Returns TVA_OK (frame fetched), TVA_EOS, TVA_ERR_TIMEOUT (no frame right + * now, session still usable) or an error. Format blocks are consumed + * internally and update tva_session_format(). + */ +int tva_session_next_frame(struct tva_session *s, struct tva_frame *out, + int timeout_ms); + +/* Latest format block; valid for the session lifetime. */ +const struct tva_format *tva_session_format(const struct tva_session *s); + +/* Human-readable reason of the last error; never NULL. */ +const char *tva_session_last_error(const struct tva_session *s); + +/* Latest error code (TVA_ERR_*), 0 when none. */ +int tva_session_last_error_code(const struct tva_session *s); + +/* Effective transport: TVA_XFER_INLINE / TVA_XFER_SHM. Requesting SHM may + * still be silently downgraded, so verify with this. */ +int tva_session_xfer_mode(const struct tva_session *s); + +/* Statistics: units sent / frames received. */ +uint64_t tva_session_units_sent(const struct tva_session *s); +uint64_t tva_session_frames_received(const struct tva_session *s); + +#ifdef __cplusplus +} +#endif + +#endif /* TVA_CLIENT_H */ diff --git a/src/gallium/frontends/va/tva_protocol.h b/src/gallium/frontends/va/tva_protocol.h new file mode 100644 index 000000000000..f41dc7c13cb8 --- /dev/null +++ b/src/gallium/frontends/va/tva_protocol.h @@ -0,0 +1,168 @@ +/* + * SPDX-License-Identifier: GPL-3.0-or-later + * + * tva_protocol.h - Wire protocol constants shared by the termux-va daemon + * (Termux side) and the Mesa termux-va bridge (container + * side, mesa-for-android-container branch + * test/add-va-bridge, file src/gallium/frontends/va/tva_protocol.h). + * + * This file is the single source of truth for everything that travels on + * the wire or must agree on both ends of the Unix socket. The two copies + * MUST be byte-identical; scripts/check-mirror.sh enforces that with cmp(1) + * (same discipline as anland-termux AGENTS.md). + * + * Copyright (C) 2026 lfdevs + * + * This program is free software: you can redistribute it and/or modify + * it under the terms of the GNU General Public License as published by + * the Free Software Foundation, version 3 of the License. + * + * This file is based on droidspaces-media-decode (src/decode-daemon.c and + * vaapi-driver/src/dmd_client.{c,h}), which is licensed under the Apache + * License, Version 2.0. It has been MODIFIED for the termux-va project and + * relicensed under the GNU General Public License version 3: + * - protocol constants extracted from the daemon/client inline defines + * into this shared header (values unchanged, wire format unchanged), + * - default socket name changed from decode.sock to termux-va.sock, + * - XFER_TCP renamed to XFER_INLINE (wire value 0 unchanged; the transport + * is always a path-based Unix socket in termux-va, TCP was removed). + * + * Protocol compatibility statement: the wire format stays byte-compatible + * with droidspaces-media-decode protocol v3 (HELLO_MAGIC 0x444D4400) so the + * upstream regression tools (tools/test_decode.py, dmd-probe) can be reused + * unchanged. + */ +#ifndef TVA_PROTOCOL_H +#define TVA_PROTOCOL_H + +#include + +#ifdef __cplusplus +extern "C" { +#endif + +/* ------------------------------------------------------------------ paths */ + +/* Fixed socket file name inside the socket directory. Both the daemon + * (directory mode) and the Mesa bridge (default endpoint) join this name + * onto the socket directory. */ +#define TVA_SOCK_NAME "termux-va.sock" + +/* Name of the socket directory relative to the shared tmp directory. + * Termux default: $TMPDIR/termux-va/termux-va.sock + * Container view: /tmp/termux-va/termux-va.sock (proot --shared-tmp) */ +#define TVA_SOCKET_DIR_NAME "termux-va" + +/* --------------------------------------------------------------- handshake */ + +/* + * Magic of the 24-byte big-endian handshake request + * [u32 magic][u32 version][u32 codec][u32 width][u32 height][u32 xfer] + * + * Kept identical to upstream droidspaces-media-decode ("DMD\0") so the + * upstream protocol v3 regression tools work unchanged. A legal NALU + * length can never equal this value (1145389568 >> MAX_FRAME), which is + * how the two are told apart... but note the handshake is REQUIRED here: + * a first word that is not the magic is rejected (status=4). + */ +#define HELLO_MAGIC 0x444D4400u + +/* + * Current protocol version. + * v2: shared-memory transport negotiation added + * v3: response may carry the endpoint dev/ino extension (bit31 of the + * name-length word), used by the client to verify it connected to + * the live socket inode (bind-mount safety). + * + * Version negotiation: the daemon accepts versions 2..HELLO_VERSION and + * takes the minimum; clients must do the same. + */ +#define HELLO_VERSION 3 +#define TVA_VERSION_MIN 2 + +/* + * Codec identifiers. This is part of the wire protocol (handshake word 3). + * Values may only be APPENDED, never reordered or reused. + */ +typedef enum { + CODEC_H264 = 0, + CODEC_HEVC = 1, + CODEC_VP9 = 2, + CODEC_VP8 = 3, + CODEC_AV1 = 4, /* AV1 Main decode */ + CODEC_MAX +} CodecId; + +/* + * Frame return transport requested in the handshake and granted in the + * response ("actual xfer" word). + * + * 0 = inline : frame data is sent over the control socket itself + * 1 = SHM : frame data is written into a memfd slot pool; the socket + * only carries a 24-byte control message naming the slot + * + * TVA: upstream calls value 0 "XFER_TCP" because upstream also had a TCP + * control transport. termux-va is Unix-socket-only, so the constant is + * renamed XFER_INLINE; the WIRE VALUE 0 IS UNCHANGED and must stay 0. + */ +typedef enum { + XFER_INLINE = 0, + XFER_SHM = 1 +} XferMode; + +/* Handshake status codes (response word 1): + * 0 accepted / 1 version / 2 codec / 3 resolution out of range / + * 4 handshake missing. Error responses are always a bare 12 bytes. */ + +/* ------------------------------------------------------------------ limits */ + +/* Hard cap of one input unit (4-byte length prefix + data). */ +#define MAX_FRAME (8 * 1024 * 1024) + +/* Max concurrent decode sessions the daemon accepts (hardware supports 16). */ +#define MAX_CLIENTS 8 + +/* ------------------------------------------------- format descriptor block */ + +/* Number of u32 words in the 32-byte format descriptor body: + * [buf_w][buf_h][stride][slice_height][crop_l][crop_t][crop_r][crop_b] */ +#define FMTDESC_WORDS 8 + +/* frame_size values that mean "this is a control message, not a frame". */ +#define FMTDESC_SENTINEL 0xFFFFFFFFu /* followed by the 32-byte format block */ +#define SHMFRAME_SENTINEL 0xFFFFFFFEu /* followed by [slot][len][pts] in one + * 24-byte SHM control message */ + +/* Capability flag in format-block header word 2: every frame header carries + * a 4th field = input unit index (round-tripped through MediaCodec PTS). */ +#define CAP_FRAME_PTS 0x00000001u + +/* + * Input unit index -> presentationTimeUs multiplier. + * + * The decoder quantizes PTS to milliseconds; feeding the raw unit index + * (1us steps) collapses everything to 0. x1000 keeps indices unique after + * quantization. The client divides by this to recover the index. + */ +#define PTS_UNIT_SCALE 1000 + +/* ------------------------------------------------------- shared memory pool */ + +/* Pool layout: [control area SHM_CTRL_BYTES][slot 0]..[slot SHM_SLOTS-1]. + * Each slot has a u32 state word in the control area: daemon sets 1 + * (release) after writing, client resets 0 (acquire) after consuming. */ +#define SHM_SLOTS 8 +#define SHM_CTRL_BYTES 4096 + +/* + * How long the daemon spins waiting for a free slot (milliseconds). + * MUST stay well above the bridge-side per-call frame timeout (5s) or the + * daemon kills sessions the client would have completed. 15s = 3x. + */ +#define SHM_SLOT_WAIT_MS 15000 + +#ifdef __cplusplus +} +#endif + +#endif /* TVA_PROTOCOL_H */ diff --git a/src/gallium/frontends/va/va_private.h b/src/gallium/frontends/va/va_private.h index f27c9085f26a..585a0d75989d 100644 --- a/src/gallium/frontends/va/va_private.h +++ b/src/gallium/frontends/va/va_private.h @@ -43,6 +43,7 @@ #include "util/u_dynarray.h" #include "util/u_thread.h" #include "util/detect_os.h" +#include "util/set.h" #if DETECT_OS_WINDOWS #define VA_PUBLIC_API @@ -341,6 +342,7 @@ typedef struct { struct pipe_context *pipe; struct pipe_context *pipe2; struct handle_table *htab; + struct set *surfaces; struct pipe_video_codec *proc; mtx_t mutex; char vendor_string[256]; @@ -436,6 +438,13 @@ typedef struct vlVaSurface { vlVaBuffer *coded_buf; struct pipe_fence_handle *fence; /* pipe_video_codec fence */ struct pipe_fence_handle *pipe_fence; /* pipe_context fence */ + /* A PRIME import may alias a decoder surface exported earlier by this + * VA display. The producer is resolved transiently before VPP samples it. */ + struct vlVaSurface *sync_surface; + /* Fence snapshot taken when Chromium imports the PRIME descriptor. The + * decoder may recycle the same dma-buf for a newer frame before VPP runs. */ + struct pipe_fence_handle *prime_fence; + bool is_prime_import; bool is_dpb; unsigned int strides[3]; unsigned int offsets[3]; @@ -562,6 +571,15 @@ MESAPROC VAStatus vlVaHandleVAProcPipelineParameterBufferType(vlVaDriver *drv, v VAStatus vlVaHandleSurfaceAllocate(vlVaDriver *drv, vlVaSurface *surface, const uint64_t *modifiers, unsigned modifiers_count); struct pipe_video_buffer *vlVaGetSurfaceBuffer(vlVaDriver *drv, vlVaSurface *surface); void vlVaSurfaceFlush(vlVaDriver *drv, vlVaSurface *surf); +/* The caller holds drv->mutex. The helper temporarily drops it while + * waiting for the producer context, then reacquires it before returning. */ +VAStatus vlVaSyncSurfaceObjectLocked(vlVaDriver *drv, vlVaSurface *surf, + uint64_t timeout_ns); +/* PRoot/KGSL has no DRM render node and cannot make progress while Chromium + * waits for a decoder fence on its single VA thread. */ +bool vlVaSurfaceNoWait(void); +vlVaSurface *surface_find_prime_producer_for_surface(vlVaDriver *drv, + vlVaSurface *import_surface); void vlVaAddRawHeader(struct util_dynarray *headers, uint8_t type, uint32_t size, uint8_t *buf, bool is_slice, uint32_t emulation_bytes_start); void vlVaGetBufferFeedback(vlVaBuffer *buf); diff --git a/src/gallium/targets/dri/meson.build b/src/gallium/targets/dri/meson.build index c526ffef4a1b..3bff01595e9e 100644 --- a/src/gallium/targets/dri/meson.build +++ b/src/gallium/targets/dri/meson.build @@ -75,7 +75,8 @@ if with_gallium_va [with_gallium_radeonsi, 'radeonsi'], [with_gallium_nouveau, 'nouveau'], [with_gallium_virgl, 'virtio_gpu'], - [with_gallium_d3d12_video, 'd3d12']] + [with_gallium_d3d12_video, 'd3d12'], + [with_termux_va_bridge, 'termuxva']] if d[0] name = '@0@_drv_video.@1@'.format(d[1], libname_suffix) va_drivers += name diff --git a/src/gbm/backends/dri/gbm_dri.c b/src/gbm/backends/dri/gbm_dri.c index 6640f16da77c..2922fdcf0c19 100644 --- a/src/gbm/backends/dri/gbm_dri.c +++ b/src/gbm/backends/dri/gbm_dri.c @@ -325,6 +325,7 @@ static const struct gbm_dri_visual gbm_dri_visuals_table[] = { { GBM_FORMAT_R16, PIPE_FORMAT_R16_UNORM }, { GBM_FORMAT_GR88, PIPE_FORMAT_R8G8_UNORM }, { GBM_FORMAT_GR1616, PIPE_FORMAT_R16G16_UNORM }, + { GBM_FORMAT_NV12, PIPE_FORMAT_R8_G8B8_420_UNORM }, { GBM_FORMAT_ARGB1555, PIPE_FORMAT_B5G5R5A1_UNORM }, { GBM_FORMAT_RGB565, PIPE_FORMAT_B5G6R5_UNORM }, { GBM_FORMAT_BGRX8888, PIPE_FORMAT_X8R8G8B8_UNORM }, @@ -379,7 +380,11 @@ gbm_dri_is_format_supported(struct gbm_device *gbm, return 0; format = core->v0.format_canonicalize(format); - if (gbm_format_to_pipe_format(format) == 0) + int pipe_format = gbm_format_to_pipe_format(format); + if (getenv("DMD_VA_LOG") && format == GBM_FORMAT_NV12) + fprintf(stderr, "gbm NV12 pipe_format=%d dmabuf=%d\\n", + pipe_format, dri->has_dmabuf_import); + if (pipe_format == 0) return 0; /* If there is no query, fall back to the small table which was originally @@ -397,8 +402,13 @@ gbm_dri_is_format_supported(struct gbm_device *gbm, /* This returns false if the format isn't supported */ if (!dri_query_dma_buf_modifiers(dri->screen, format, 0, NULL, NULL, - &count)) + &count)) { + if (getenv("DMD_VA_LOG") && format == GBM_FORMAT_NV12) + fprintf(stderr, "gbm NV12 dma-buf query rejected\\n"); return 0; + } + if (getenv("DMD_VA_LOG") && format == GBM_FORMAT_NV12) + fprintf(stderr, "gbm NV12 dma-buf query count=%d\\n", count); return 1; } @@ -899,6 +909,10 @@ gbm_dri_bo_create(struct gbm_device *gbm, format = core->v0.format_canonicalize(format); + if (getenv("DMD_VA_LOG") && format == GBM_FORMAT_NV12) + fprintf(stderr, "gbm NV12 create usage=%#x export=%d modifiers=%p count=%u\n", + usage, dri->has_dmabuf_export, (const void *) modifiers, count); + if (usage & GBM_BO_USE_WRITE || !dri->has_dmabuf_export) return create_dumb(gbm, width, height, format, usage); @@ -912,6 +926,9 @@ gbm_dri_bo_create(struct gbm_device *gbm, bo->base.v0.format = format; pipe_format = gbm_format_to_pipe_format(format); + if (getenv("DMD_VA_LOG") && format == GBM_FORMAT_NV12) + fprintf(stderr, "gbm NV12 pipe_format=%d dri_use=%#x size=%ux%u\n", + pipe_format, dri_use, width, height); if (pipe_format == 0) { errno = EINVAL; goto failed; @@ -1017,8 +1034,11 @@ gbm_dri_bo_create(struct gbm_device *gbm, mods_filtered ? mods_filtered : modifiers, mods_filtered ? count_filtered : count, bo); - if (bo->image == NULL) + if (bo->image == NULL) { + if (getenv("DMD_VA_LOG") && format == GBM_FORMAT_NV12) + fprintf(stderr, "gbm NV12 image creation failed errno=%d\n", errno); goto failed; + } free(mods_filtered); mods_filtered = NULL; diff --git a/src/glx/glxext.c b/src/glx/glxext.c index 52f29ba79fd1..2d04191059cc 100644 --- a/src/glx/glxext.c +++ b/src/glx/glxext.c @@ -17,6 +17,7 @@ #include #include #include +#include #include "glxclient.h" #include @@ -979,6 +980,11 @@ __glXInitialize(Display * dpy) enum glx_driver glx_driver = 0; const char *env = os_get_option("MESA_LOADER_DRIVER_OVERRIDE"); + const char *termux_backend = os_get_option("TERMUX_VA_GPU_BACKEND"); + const bool termux_kgsl = termux_backend && + !strcmp(termux_backend, "kgsl"); + if (!env && termux_kgsl && setenv("MESA_LOADER_DRIVER_OVERRIDE", "kgsl", 0) == 0) + env = os_get_option("MESA_LOADER_DRIVER_OVERRIDE"); #if defined(GLX_DIRECT_RENDERING) Bool glx_direct = !debug_get_bool_option("LIBGL_ALWAYS_INDIRECT", false); @@ -1069,7 +1075,8 @@ __glXInitialize(Display * dpy) #endif #endif /* GLX_DIRECT_RENDERING */ - if (!AllocAndFetchScreenConfigs(dpy, dpyPriv, glx_driver, !env)) { + if (!AllocAndFetchScreenConfigs(dpy, dpyPriv, glx_driver, + !env && !termux_kgsl)) { Bool fail = True; #if defined(GLX_DIRECT_RENDERING) if (glx_driver & GLX_DRIVER_ZINK_INFER) { diff --git a/src/loader/loader.c b/src/loader/loader.c index 939ae54965a2..59d5d417a93d 100644 --- a/src/loader/loader.c +++ b/src/loader/loader.c @@ -795,6 +795,22 @@ loader_get_driver_for_fd(int fd) const char *override = os_get_option("MESA_LOADER_DRIVER_OVERRIDE"); if (override && strlen(override)) return strdup(override); + + /* A PRoot Android container can expose KGSL without a DRM node. The + * explicit bridge backend is the equivalent of + * MESA_LOADER_DRIVER_OVERRIDE=kgsl for that device, but keep the + * selection limited to an actual KGSL descriptor so unrelated DRM + * fds are still inferred normally. */ + const char *backend = os_get_option("TERMUX_VA_GPU_BACKEND"); + if (backend && !strcmp(backend, "kgsl")) { +#ifdef __linux__ + struct stat fd_stat, kgsl_stat; + if (fstat(fd, &fd_stat) == 0 && + stat("/dev/kgsl-3d0", &kgsl_stat) == 0 && + fd_stat.st_rdev == kgsl_stat.st_rdev) + return strdup("kgsl"); +#endif + } } #if defined(USE_DRICONF) diff --git a/src/meson.build b/src/meson.build index c36f44b13978..2899fb70b74a 100644 --- a/src/meson.build +++ b/src/meson.build @@ -61,6 +61,9 @@ endif if with_gallium_or_lvp or with_gbm or with_platform_wayland or with_platform_x11 subdir('loader') endif +if with_termux_va_wayland_shim + subdir('termux-va') +endif subdir('compiler') if with_poly subdir('poly') diff --git a/src/termux-va/meson.build b/src/termux-va/meson.build new file mode 100644 index 000000000000..83ec5c588426 --- /dev/null +++ b/src/termux-va/meson.build @@ -0,0 +1,10 @@ +# SPDX-License-Identifier: MIT + +libtva_drm_shim_wayland = shared_library( + 'tva_drm_shim_wayland', + 'tva_drm_shim_wayland.c', + dependencies : [dep_libdrm, dep_dl], + gnu_symbol_visibility : 'default', + install : true, + install_tag : 'runtime', +) diff --git a/src/termux-va/tva_drm_shim_wayland.c b/src/termux-va/tva_drm_shim_wayland.c new file mode 100644 index 000000000000..ec28a396d8af --- /dev/null +++ b/src/termux-va/tva_drm_shim_wayland.c @@ -0,0 +1,498 @@ +/* + * Copyright © 2026 lfdevs + * SPDX-License-Identifier: MIT + * + * This compatibility shim lets Chromium's native Wayland backend use the + * KGSL device exposed by a DRM-less PRoot container. It is intended to be + * loaded into Chromium with LD_PRELOAD, not installed as a system libdrm + * replacement. + */ + +#ifndef _GNU_SOURCE +#define _GNU_SOURCE +#endif + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include + +static const char *const kgsl_path = "/dev/kgsl-3d0"; + +#if defined(__GLIBC__) || defined(__BIONIC__) +typedef unsigned long tva_ioctl_request_t; +#else +/* musl declares ioctl(2)'s request argument as int. */ +typedef int tva_ioctl_request_t; +#endif + +static drmVersionPtr (*real_drmGetVersion)(int); +static void (*real_drmFreeVersion)(drmVersionPtr); +static int (*real_drmGetNodeTypeFromFd)(int); +static int (*real_drmGetDevices2)(uint32_t, drmDevicePtr[], int); +static int (*real_drmGetDevice2)(int, uint32_t, drmDevicePtr *); +static int (*real_drmGetDevice)(int, drmDevicePtr *); +static void (*real_drmFreeDevice)(drmDevicePtr *); +static void (*real_drmFreeDevices)(drmDevicePtr[], int); +static int (*real_drmGetDeviceFromDevId)(dev_t, uint32_t, drmDevicePtr *); +static int (*real_drmGetNodeTypeFromDevId)(dev_t); +static int (*real_drmIoctl)(int, unsigned long, void *); +static int (*real_ioctl)(int, tva_ioctl_request_t, ...); +static char *(*real_drmGetDeviceNameFromFd2)(int); +static char *(*real_drmGetDeviceNameFromFd)(int); +static char *(*real_drmGetRenderDeviceNameFromFd)(int); +static char *(*real_drmGetPrimaryDeviceNameFromFd)(int); + +static bool is_kgsl_fd(int fd); + +static void +init_real(void) +{ +#define LOAD(name) \ + do { \ + if (!real_##name) \ + real_##name = dlsym(RTLD_NEXT, #name); \ + } while (0) + LOAD(drmGetVersion); + LOAD(drmFreeVersion); + LOAD(drmGetNodeTypeFromFd); + LOAD(drmGetDevices2); + LOAD(drmGetDevice2); + LOAD(drmGetDevice); + LOAD(drmFreeDevice); + LOAD(drmFreeDevices); + LOAD(drmGetDeviceFromDevId); + LOAD(drmGetNodeTypeFromDevId); + LOAD(drmIoctl); + LOAD(drmGetDeviceNameFromFd2); + LOAD(drmGetDeviceNameFromFd); + LOAD(drmGetRenderDeviceNameFromFd); + LOAD(drmGetPrimaryDeviceNameFromFd); +#undef LOAD +} + +static void +fill_fake_drm_version(struct drm_version *version) +{ + int name_capacity = version->name_len; + + version->version_major = 1; + version->version_minor = 0; + version->version_patchlevel = 0; + version->name_len = 4; + version->date_len = 0; + version->desc_len = 0; + if (version->name && name_capacity >= 4) + memcpy(version->name, "kgsl", 4); +} + +/* GBM probes a render node with DRM_IOCTL_GET_CAP before it creates a + * native-pixmap buffer. KGSL is not a DRM device and returns ENOTTY for this + * request, but the Mesa KGSL driver already provides the actual allocation + * path. Report only the capability answers needed by the userspace probe; + * do not claim PRIME or modifier support that KGSL cannot provide. */ +static bool +fill_fake_drm_cap(struct drm_get_cap *cap) +{ + if (!cap) + return false; + + switch (cap->capability) { + case DRM_CAP_TIMESTAMP_MONOTONIC: + cap->value = 1; + break; + case DRM_CAP_PRIME: + cap->value = DRM_PRIME_CAP_IMPORT | DRM_PRIME_CAP_EXPORT; + break; + case DRM_CAP_DUMB_BUFFER: + case DRM_CAP_ADDFB2_MODIFIERS: + cap->value = 0; + break; + default: + cap->value = 0; + break; + } + return true; +} + +/* Chromium's Wayland Ozone code issues this ioctl directly instead of going + * through libdrm. KGSL is not a DRM device, so provide only the version + * response needed by its render-node handle validation. */ +int +ioctl(int fd, tva_ioctl_request_t request, ...) +{ + init_real(); + + va_list ap; + va_start(ap, request); + void *arg = va_arg(ap, void *); + va_end(ap); + + if (getenv("TERMUX_VA_DRM_SHIM_LOG") && is_kgsl_fd(fd)) + fprintf(stderr, "tva-drm-shim: ioctl fd=%d request=0x%lx arg=%p\n", + fd, (unsigned long)request, arg); + + if (is_kgsl_fd(fd) && + request == (tva_ioctl_request_t) DRM_IOCTL_VERSION && arg) { + fill_fake_drm_version(arg); + if (getenv("TERMUX_VA_DRM_SHIM_LOG")) + fprintf(stderr, "tva-drm-shim: ioctl DRM_IOCTL_VERSION fd=%d\n", fd); + return 0; + } + + if (is_kgsl_fd(fd) && + request == (tva_ioctl_request_t) DRM_IOCTL_GET_CAP && arg && + fill_fake_drm_cap(arg)) { + if (getenv("TERMUX_VA_DRM_SHIM_LOG")) + fprintf(stderr, "tva-drm-shim: ioctl DRM_IOCTL_GET_CAP fd=%d cap=%llu value=%llu\n", + fd, (unsigned long long)((struct drm_get_cap *)arg)->capability, + (unsigned long long)((struct drm_get_cap *)arg)->value); + return 0; + } + + if (!real_ioctl) + real_ioctl = dlsym(RTLD_NEXT, "ioctl"); + if (real_ioctl) + return real_ioctl(fd, request, arg); + + errno = ENOSYS; + return -1; +} + +static bool +is_kgsl_fd(int fd) +{ + struct stat fd_st; + struct stat kgsl_st; + + return fstat(fd, &fd_st) == 0 && stat(kgsl_path, &kgsl_st) == 0 && + S_ISCHR(fd_st.st_mode) && fd_st.st_rdev == kgsl_st.st_rdev; +} + +static bool +is_kgsl_dev(dev_t dev) +{ + struct stat st; + + return stat(kgsl_path, &st) == 0 && st.st_rdev == dev; +} + +int +drmIoctl(int fd, unsigned long request, void *arg) +{ + init_real(); + if (getenv("TERMUX_VA_DRM_SHIM_LOG") && is_kgsl_fd(fd)) + fprintf(stderr, "tva-drm-shim: drmIoctl fd=%d request=0x%lx arg=%p\n", + fd, request, arg); + if (getenv("TERMUX_VA_DRM_SHIM_LOG") && is_kgsl_fd(fd) && + request == DRM_IOCTL_GET_CAP) + fprintf(stderr, "tva-drm-shim: GET_CAP matched expected=0x%lx\n", + (unsigned long)DRM_IOCTL_GET_CAP); + if (is_kgsl_fd(fd) && request == DRM_IOCTL_VERSION && arg) { + fill_fake_drm_version(arg); + return 0; + } + + if (is_kgsl_fd(fd) && request == DRM_IOCTL_GET_CAP && arg && + fill_fake_drm_cap(arg)) { + if (getenv("TERMUX_VA_DRM_SHIM_LOG")) + fprintf(stderr, "tva-drm-shim: drmIoctl GET_CAP cap=%llu value=%llu\n", + (unsigned long long)((struct drm_get_cap *)arg)->capability, + (unsigned long long)((struct drm_get_cap *)arg)->value); + return 0; + } + + if (real_drmIoctl) + return real_drmIoctl(fd, request, arg); + + errno = ENOSYS; + return -1; +} + +static drmDevicePtr +make_fake_device(void) +{ + drmDevicePtr device = calloc(1, sizeof(*device)); + if (!device) + return NULL; + + device->nodes = calloc(DRM_NODE_MAX, sizeof(*device->nodes)); + device->businfo.pci = calloc(1, sizeof(*device->businfo.pci)); + device->deviceinfo.pci = calloc(1, sizeof(*device->deviceinfo.pci)); + if (!device->nodes || !device->businfo.pci || !device->deviceinfo.pci) + goto fail; + + device->nodes[DRM_NODE_RENDER] = strdup(kgsl_path); + device->nodes[DRM_NODE_PRIMARY] = strdup(kgsl_path); + if (!device->nodes[DRM_NODE_RENDER] || !device->nodes[DRM_NODE_PRIMARY]) + goto fail; + + /* Chromium's VA-API discovery currently filters out non-PCI devices. + * Keep the KGSL path while presenting the same neutral PCI identity that + * the container's ANGLE setup uses. */ + device->businfo.pci->domain = 0; + device->businfo.pci->bus = 0; + device->businfo.pci->dev = 0; + device->businfo.pci->func = 0; + device->deviceinfo.pci->vendor_id = 0; + device->deviceinfo.pci->device_id = 0; + device->available_nodes = (1 << DRM_NODE_RENDER) | (1 << DRM_NODE_PRIMARY); + device->bustype = DRM_BUS_PCI; + return device; + +fail: + free(device->deviceinfo.pci); + free(device->businfo.pci); + if (device->nodes) { + free(device->nodes[DRM_NODE_RENDER]); + free(device->nodes[DRM_NODE_PRIMARY]); + free(device->nodes); + } + free(device); + return NULL; +} + +static bool +is_fake_device(drmDevicePtr device) +{ + return device && device->nodes && device->nodes[DRM_NODE_RENDER] && + strcmp(device->nodes[DRM_NODE_RENDER], kgsl_path) == 0; +} + +static void +free_fake_device(drmDevicePtr device) +{ + if (!device) + return; + + free(device->deviceinfo.pci); + free(device->businfo.pci); + if (device->nodes) { + free(device->nodes[DRM_NODE_RENDER]); + free(device->nodes[DRM_NODE_PRIMARY]); + free(device->nodes); + } + free(device); +} + +static int +fake_device_result(drmDevicePtr *out) +{ + if (!out) + return -EINVAL; + + *out = make_fake_device(); + if (!*out) + return -ENOMEM; + + if (getenv("TERMUX_VA_DRM_SHIM_LOG")) + fprintf(stderr, "tva-drm-shim: exposing %s as DRM render node\n", + kgsl_path); + return 0; +} + +int +drmGetDevices2(uint32_t flags, drmDevicePtr devices[], int max_devices) +{ + (void)flags; + init_real(); + + /* Chromium's GPU process may not preserve all application environment + * variables. The presence of the KGSL node is therefore the opt-in for + * this process-local compatibility path. */ + struct stat st; + int stat_rc = stat(kgsl_path, &st); + if (getenv("TERMUX_VA_DRM_SHIM_LOG")) + fprintf(stderr, "tva-drm-shim: drmGetDevices2 stat=%d errno=%d max=%d\n", + stat_rc, errno, max_devices); + if (stat_rc != 0) { + if (real_drmGetDevices2) + return real_drmGetDevices2(flags, devices, max_devices); + errno = ENOSYS; + return -1; + } + + if (!devices || max_devices == 0) + return 1; + if (max_devices < 1) + return 0; + + devices[0] = make_fake_device(); + return devices[0] ? 1 : -ENOMEM; +} + +int +drmGetDevices(drmDevicePtr devices[], int max_devices) +{ + return drmGetDevices2(0, devices, max_devices); +} + +void +drmFreeDevices(drmDevicePtr devices[], int count) +{ + init_real(); + for (int i = 0; i < count; i++) { + if (!devices || !devices[i]) + continue; + if (is_fake_device(devices[i])) + free_fake_device(devices[i]); + else if (real_drmFreeDevice) + real_drmFreeDevice(&devices[i]); + } +} + +void +drmFreeDevice(drmDevicePtr *device) +{ + init_real(); + if (device && is_fake_device(*device)) { + free_fake_device(*device); + *device = NULL; + return; + } + + if (real_drmFreeDevice) + real_drmFreeDevice(device); +} + +int +drmGetDeviceFromDevId(dev_t dev_id, uint32_t flags, drmDevicePtr *device) +{ + init_real(); + if (is_kgsl_dev(dev_id)) + return fake_device_result(device); + if (real_drmGetDeviceFromDevId) + return real_drmGetDeviceFromDevId(dev_id, flags, device); + errno = ENOSYS; + return -1; +} + +int +drmGetNodeTypeFromDevId(dev_t dev_id) +{ + init_real(); + if (is_kgsl_dev(dev_id)) + return DRM_NODE_RENDER; + if (real_drmGetNodeTypeFromDevId) + return real_drmGetNodeTypeFromDevId(dev_id); + errno = ENOSYS; + return -1; +} + +int +drmGetDevice2(int fd, uint32_t flags, drmDevicePtr *device) +{ + init_real(); + if (is_kgsl_fd(fd)) + return fake_device_result(device); + if (real_drmGetDevice2) + return real_drmGetDevice2(fd, flags, device); + errno = ENOSYS; + return -1; +} + +int +drmGetDevice(int fd, drmDevicePtr *device) +{ + return drmGetDevice2(fd, 0, device); +} + +int +drmGetNodeTypeFromFd(int fd) +{ + init_real(); + if (is_kgsl_fd(fd)) + return DRM_NODE_RENDER; + if (real_drmGetNodeTypeFromFd) + return real_drmGetNodeTypeFromFd(fd); + errno = ENOSYS; + return -1; +} + +static char * +device_name(int fd) +{ + return is_kgsl_fd(fd) ? strdup(kgsl_path) : NULL; +} + +char * +drmGetDeviceNameFromFd2(int fd) +{ + init_real(); + char *name = device_name(fd); + return name ? name : (real_drmGetDeviceNameFromFd2 ? + real_drmGetDeviceNameFromFd2(fd) : NULL); +} + +char * +drmGetDeviceNameFromFd(int fd) +{ + init_real(); + char *name = device_name(fd); + return name ? name : (real_drmGetDeviceNameFromFd ? + real_drmGetDeviceNameFromFd(fd) : NULL); +} + +char * +drmGetRenderDeviceNameFromFd(int fd) +{ + init_real(); + char *name = device_name(fd); + return name ? name : (real_drmGetRenderDeviceNameFromFd ? + real_drmGetRenderDeviceNameFromFd(fd) : NULL); +} + +char * +drmGetPrimaryDeviceNameFromFd(int fd) +{ + init_real(); + char *name = device_name(fd); + return name ? name : (real_drmGetPrimaryDeviceNameFromFd ? + real_drmGetPrimaryDeviceNameFromFd(fd) : NULL); +} + +drmVersionPtr +drmGetVersion(int fd) +{ + init_real(); + if (!is_kgsl_fd(fd)) + return real_drmGetVersion ? real_drmGetVersion(fd) : NULL; + + drmVersionPtr version = calloc(1, sizeof(*version)); + if (!version) + return NULL; + + version->version_major = 1; + version->name_len = 4; + version->name = strdup("kgsl"); + if (!version->name) { + free(version); + return NULL; + } + return version; +} + +void +drmFreeVersion(drmVersionPtr version) +{ + init_real(); + if (version && version->name && version->name_len == 4 && + memcmp(version->name, "kgsl", 4) == 0) { + free(version->name); + free(version); + return; + } + + if (real_drmFreeVersion) + real_drmFreeVersion(version); +} diff --git a/src/x11/x11_dri3.c b/src/x11/x11_dri3.c index b995372be661..45badc90c829 100644 --- a/src/x11/x11_dri3.c +++ b/src/x11/x11_dri3.c @@ -50,8 +50,13 @@ x11_dri3_open(xcb_connection_t *conn, const xcb_query_extension_reply_t *extension; const char *env = getenv("MESA_LOADER_DRIVER_OVERRIDE"); - if (env && !strcmp(env, "kgsl")) - return open("/dev/kgsl-3d0", O_RDWR); + const char *backend = getenv("TERMUX_VA_GPU_BACKEND"); + if ((env && !strcmp(env, "kgsl")) || + (backend && !strcmp(backend, "kgsl"))) { + fd = open("/dev/kgsl-3d0", O_RDWR | O_CLOEXEC); + if (fd >= 0) + return fd; + } xcb_prefetch_extension_data(conn, &xcb_dri3_id); extension = xcb_get_extension_data(conn, &xcb_dri3_id);